driftproof 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/skill.js CHANGED
@@ -59,7 +59,13 @@ function loadSkill(skillDir) {
59
59
  }
60
60
  const suiteRaw = JSON.parse(fs.readFileSync(suitePath, 'utf8'));
61
61
  const cases = normalizeCases(suiteRaw);
62
- const suiteHash = sha256Canonical(cases);
62
+ // suite_hash is over the CORE case fields only ({id, prompt, rubric,
63
+ // pass_threshold}); optional annotations (checks, and the claim/grounding the
64
+ // gate reads straight from the raw suite) are excluded so adding them never
65
+ // disturbs a receipt's suite_hash. See spec/RECEIPT.md § suite.
66
+ const suiteHash = sha256Canonical(cases.map((c) => ({
67
+ id: c.id, prompt: c.prompt, rubric: c.rubric, pass_threshold: c.pass_threshold,
68
+ })));
63
69
 
64
70
  return {
65
71
  dir,
@@ -110,7 +116,11 @@ function normalizeCases(raw) {
110
116
  if (!rubric) throw new Error(`case "${id}" is missing a rubric/criteria/expected`);
111
117
  const threshold = typeof c.pass_threshold === 'number' ? c.pass_threshold
112
118
  : typeof c.threshold === 'number' ? c.threshold : 0.7;
113
- return { id, prompt: String(prompt), rubric: String(rubric), pass_threshold: threshold };
119
+ const norm = { id, prompt: String(prompt), rubric: String(rubric), pass_threshold: threshold };
120
+ // Optional deterministic post-checks (v0.3.1). Carried through for the runner
121
+ // but EXCLUDED from suite_hash (see loadSkill) so they never disturb a receipt.
122
+ if (Array.isArray(c.checks) && c.checks.length) norm.checks = c.checks;
123
+ return norm;
114
124
  });
115
125
  }
116
126
 
@@ -0,0 +1,31 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Value-per-token axis.
5
+ //
6
+ // A skill is not free: its SKILL.md is prepended to every generation, so a skill
7
+ // that lifts scores by +0.10 for 400 tokens is a very different proposition from
8
+ // one that lifts +0.10 for 6,000 tokens. Driftproof reports each skill's lift
9
+ // (delta) BOTH raw AND normalized per 1,000 skill tokens.
10
+ //
11
+ // Token estimate: a coarse `ceil(chars / 4)` proxy — the standard rough rule for
12
+ // English text, NOT a model tokenizer (no tiktoken dependency; the number is a
13
+ // consistent proxy for comparing skills, not a billing figure). Documented on
14
+ // docs/methodology.html.
15
+
16
+ const CHARS_PER_TOKEN = 4;
17
+
18
+ function estimateTokens(text) {
19
+ const s = String(text || '');
20
+ if (!s.length) return 0;
21
+ return Math.ceil(s.length / CHARS_PER_TOKEN);
22
+ }
23
+
24
+ // Lift per 1,000 skill tokens: delta / (skillTokens / 1000). Null when the token
25
+ // count is unknown or zero (avoids a divide-by-zero masquerading as infinite value).
26
+ function deltaPer1kTokens(delta, skillTokens) {
27
+ if (typeof delta !== 'number' || !skillTokens || skillTokens <= 0) return null;
28
+ return Math.round((delta / (skillTokens / 1000)) * 1e4) / 1e4;
29
+ }
30
+
31
+ module.exports = { estimateTokens, deltaPer1kTokens, CHARS_PER_TOKEN };
package/lib/verdict.js CHANGED
@@ -21,6 +21,10 @@ const VERDICTS = {
21
21
  PASSED: { color: 'brightgreen', word: 'passing' },
22
22
  NO_EFFECT: { color: 'lightgrey', word: 'no effect' },
23
23
  REGRESSED: { color: 'red', word: 'regressed' },
24
+ // Interop (Phase 7): a receipt below TESTED (imported/DECLARED) or with no
25
+ // delta (a source tool with no baseline mode) never gets a pass/fail verdict
26
+ // — declared numbers are recorded, not verified, so the badge says so.
27
+ NOT_MEASURED: { color: 'lightgrey', word: 'not measured' },
24
28
  };
25
29
 
26
30
  // Strip a trailing -YYYYMMDD date stamp so the badge reads cleanly
@@ -35,8 +39,13 @@ function verdictFromReceipt(receipt) {
35
39
  const cmp = (receipt && receipt.comparison) || {};
36
40
  const delta = typeof cmp.delta === 'number' ? cmp.delta : 0;
37
41
  const model = shortModel(receipt && receipt.run && receipt.run.model_id);
42
+ // Below-TESTED receipts (imported/DECLARED) and null-delta receipts are never
43
+ // verdicted — we did not run the suite, so we do not certify the outcome.
44
+ const level = (receipt && receipt.verification_level) || 'TESTED';
45
+ const measured = level === 'TESTED' && typeof cmp.delta === 'number';
38
46
  let verdict;
39
- if (delta >= EFFECT_FLOOR) verdict = 'PASSED';
47
+ if (!measured) verdict = 'NOT_MEASURED';
48
+ else if (delta >= EFFECT_FLOOR) verdict = 'PASSED';
40
49
  else if (delta <= -EFFECT_FLOOR) verdict = 'REGRESSED';
41
50
  else verdict = 'NO_EFFECT';
42
51
  const meta = VERDICTS[verdict];
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "driftproof",
3
- "version": "0.3.0",
3
+ "version": "0.4.0",
4
4
  "description": "Continuous verification of agent skills: run a skill's eval suite with and without the skill across model versions, emit signed dated receipts, and diff receipts into drift reports.",
5
5
  "license": "Apache-2.0",
6
6
  "keywords": [
package/spec/RECEIPT.md CHANGED
@@ -1,5 +1,5 @@
1
1
  <!-- SPDX-License-Identifier: Apache-2.0 -->
2
- # Driftproof receipt — spec v0.3
2
+ # Driftproof receipt — spec v0.3.1
3
3
 
4
4
  A **receipt** is a signed, dated record of running one agent skill's eval suite
5
5
  **with** and **without** the skill on one model version, with the judge **sampled**
@@ -11,13 +11,74 @@ The machine-readable contract is [`receipt.schema.json`](./receipt.schema.json)
11
11
  (JSON Schema, draft 2020-12). This document is the human companion. Where they
12
12
  disagree, the schema wins.
13
13
 
14
- **Versioning.** The current schema is v0.3
14
+ **Versioning.** The current schema is v0.3.1
15
15
  ([`receipt.schema.json`](./receipt.schema.json)). Prior schemas are kept as
16
+ [`receipt.v0.3.schema.json`](./receipt.v0.3.schema.json),
16
17
  [`receipt.v0.2.schema.json`](./receipt.v0.2.schema.json) and
17
18
  [`receipt.v0.1.schema.json`](./receipt.v0.1.schema.json); the validator picks the
18
- schema by the receipt's own `schema_version`, so v0.1 and v0.2 receipts still load
19
- and validate. v0.3 is an **additive** bump — every field it adds is required in a
20
- fresh run, but older receipts are read unchanged against their own schema.
19
+ schema by the receipt's own `schema_version`, so v0.1, v0.2 and v0.3 receipts still
20
+ load and validate. v0.3.1 is an **additive** bump — `run.provider` is required in a
21
+ fresh run and the rest is optional, but older receipts are read unchanged against
22
+ their own schema.
23
+
24
+ ## Interop-additive revision (Phase 7 — receipts as an open format)
25
+
26
+ The v0.3.1 schema gained an **additive interop revision** so receipts can be
27
+ **imported** from neighboring eval tools with honest epistemics (see
28
+ [`docs/interop.md`](../docs/interop.md) and the site's `/interop.html`):
29
+
30
+ - `run.surface` and `run.judge.surface` gain **`"external"`** (the run happened
31
+ on another tool's harness); `run.transcripts` gains **`"none"`** (nothing
32
+ retained, not even hashes — only honest on an import).
33
+ - New optional **`run.source`** — provenance of a converted receipt, e.g.
34
+ `"imported/agent-skills-eval"` or `"imported/skillgrade"`.
35
+ - `skill.content_hash`, `suite.suite_hash`, and per-case `judge.rubric_hash` may
36
+ be **`null`**, per-case `generation_hash`/`judge_sample_hashes` may be
37
+ **omitted**, and `comparison.baseline_score`/`delta`/`delta_uncertainty` may be
38
+ **`null`** (a source tool with no baseline mode) — hashes and baselines are
39
+ **never fabricated**. `suite.format` may name a non-agentskills format (e.g.
40
+ `"skillgrade/eval.yaml"`).
41
+ - **The TESTED tightening** (a top-level schema conditional) makes every one of
42
+ those relaxations available **only below `TESTED`**: a receipt claiming
43
+ `verification_level: "TESTED"` must still carry the full evidence chain —
44
+ content/suite hashes, per-case generation + judge-sample hashes, a
45
+ non-`external` surface, numeric comparison — exactly as before. No previously
46
+ issued receipt is invalidated, and `TESTED` keeps its meaning.
47
+ - Structural consequence, enforced in code: **drift verdicts require `TESTED`
48
+ on both sides** — `diff` reports **NOT MEASURED** against a `DECLARED`
49
+ (imported) receipt, and a `DECLARED` receipt's badge reads *not measured*.
50
+
51
+ ## What changed from v0.3 (multi-provider — spec v0.3.1)
52
+
53
+ - **`run.provider`** (required) — the two-axis provider the target model ran on:
54
+ `"anthropic"` or `"openai"` (from the registry `provider`, else inferred from the
55
+ id). Driftproof is now two-provider: the same suites and the same fixed judge,
56
+ run on Claude and GPT substrates.
57
+ - **`run.surface`** gains the OpenAI lanes: the enum is now
58
+ `"api" | "claude-cli" | "openai-api" | "openai-cli"`. `openai-api` = a
59
+ Chat-Completions-compatible API (base_url configurable); `openai-cli` = the Codex
60
+ subscription surface (`codex exec`). `run.judge.surface` accepts the same set.
61
+ - **`run.surface_overhead_note`** (optional) — present on the `openai-cli` surface:
62
+ states the fixed Codex base-instruction preamble (~12–15k input tokens per call)
63
+ that the harness prepends and does not control. The model id is set by us via
64
+ `-m` (it is not echoed in the Codex JSONL stream).
65
+ - **Per-case `checks[]`** (optional) — deterministic post-check results, each
66
+ `{ name, kind, pass }` with `kind ∈ regex | contains | not_contains | min_length`.
67
+ Structural/regex assertions run on the model output **alongside** the judge and
68
+ reported as a **separate column** — **supplementary evidence only, never folded
69
+ into the `outcome`/band verdict.**
70
+ - **`skill.tokens`** (optional) — the estimated token size of the skill's SKILL.md
71
+ (a coarse `chars/4` proxy, not a model tokenizer), used for the **value-per-token**
72
+ axis: `delta` per 1k skill tokens. Method documented on the methodology page.
73
+ - **Per-case `case_status`** (optional; default `"ok"`) — `"failed_timeout"` marks a
74
+ case whose model/judge call persistently timed out after retries. Such a case is
75
+ **recorded WITHOUT fabricated samples/hashes** and is **excluded from the
76
+ aggregates** — a band is never invented from a case that did not complete. When
77
+ any case failed, the run is stamped **`run.status: "incomplete"`** (+
78
+ `run.failed_case_count`), and a drift/durability report **must exclude an
79
+ incomplete receipt from verdicts** (listing it honestly as "not measured").
80
+ - These are additive; a v0.3 or earlier receipt reads unchanged against its own
81
+ frozen schema.
21
82
 
22
83
  ## Design goals
23
84
 
@@ -60,7 +121,7 @@ fresh run, but older receipts are read unchanged against their own schema.
60
121
 
61
122
  ## Fields
62
123
 
63
- ### `schema_version` (string, required) — `"0.3"`.
124
+ ### `schema_version` (string, required) — `"0.3.1"`.
64
125
 
65
126
  ### `skill` (object, required)
66
127
  | field | type | notes |
@@ -68,6 +129,7 @@ fresh run, but older receipts are read unchanged against their own schema.
68
129
  | `name` | string | From SKILL.md front-matter or its H1. |
69
130
  | `version` | string | Skill's declared version. |
70
131
  | `content_hash` | sha256 hex | Over `SKILL.md` + every bundled file (the `evals/` dir is **excluded** — hashed separately as `suite_hash`), path + content, path-sorted. |
132
+ | `tokens` | integer | **v0.3.1, optional.** Estimated SKILL.md token size (coarse `chars/4` proxy) for the value-per-token axis. |
71
133
 
72
134
  ### `suite` (object, required)
73
135
  | field | type | notes |
@@ -81,7 +143,9 @@ fresh run, but older receipts are read unchanged against their own schema.
81
143
  |---|---|---|
82
144
  | `model_id` | string | Canonical model id run on. |
83
145
  | `model_release_date` | ISO date or `null` | Release date **if known**, else `null`. |
84
- | `surface` | `"api"` \| `"claude-cli"` | `api` = Messages API; `claude-cli` = spawned `claude -p` (subscription). |
146
+ | `provider` | `"anthropic"` \| `"openai"` | **v0.3.1.** The two-axis provider (registry `provider`, else inferred from the id). |
147
+ | `surface` | `"api"` \| `"claude-cli"` \| `"openai-api"` \| `"openai-cli"` | `api` = Anthropic Messages API; `claude-cli` = spawned `claude -p` (subscription); `openai-api` = Chat-Completions-compatible API (base_url configurable); `openai-cli` = Codex subscription (`codex exec`). |
148
+ | `surface_overhead_note` | string | **v0.3.1, optional.** On `openai-cli`: the fixed Codex base-instruction preamble (~12–15k input tokens/call) the harness prepends and does not control. |
85
149
  | `runner_version` | string | Runner version, for reproducibility. |
86
150
  | `date_utc` | ISO 8601 UTC | When the run finished. |
87
151
  | `registry` | `"registered"` \| `"unregistered"` | **v0.3.** Whether `model_id` resolved in the model registry (`config/models.json`). |
@@ -111,6 +175,7 @@ receipt says so.
111
175
  | `score` | number 0–1 | Alias of `mean` (kept for v0.1 readers). |
112
176
  | `threshold` | number or `null` | The case's pass threshold (or `null`). |
113
177
  | `reason` | string | One-line judge rationale (optional). |
178
+ | `checks` | array | **v0.3.1, optional.** Deterministic post-check results, each `{ name, kind, pass }` (`kind ∈ regex/contains/not_contains/min_length`). Run alongside the judge; a **separate column**, **not** folded into `outcome`/the band verdict. |
114
179
  | `judge` | object | `{ model_id, rubric_hash }` — who graded and a hash binding the grade to the exact rubric + judge system prompt. |
115
180
  - **`aggregates`** — `{ with_skill, baseline }`, each
116
181
  `{ case_count, pass_count, borderline_count?, mean_score, stddev }`. Here
@@ -2,14 +2,59 @@
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
3
  "$id": "https://driftproofhq.com/spec/receipt.schema.json",
4
4
  "title": "driftproof receipt",
5
- "description": "A signed, dated record of running one agent skill's eval suite with and without the skill on one model version, with sampled judge scores and confidence bands. Receipt spec v0.3 — adds transcript auditability (per-sample sha256 hashes + optional transcript retention) and registry/pricing provenance.",
5
+ "description": "A signed, dated record of running one agent skill's eval suite with and without the skill on one model version, with sampled judge scores and confidence bands. Receipt spec v0.3.1 — additive over v0.3: adds run.provider (two-axis provider), expands the surface enum with the OpenAI lanes (openai-api, openai-cli), adds an optional run.surface_overhead_note (fixed harness preamble on the openai/cli surface), optional per-case deterministic post-check results, and an optional skill.tokens count for the value-per-token axis. Interop-additive revision: receipts IMPORTED from external tools (surface 'external', run.source 'imported/<tool>', verification_level DECLARED) may carry null content/suite/rubric hashes and omit generation hashes — hashes are never fabricated; the TESTED tightening (see allOf) requires the full evidence chain whenever verification_level is TESTED, so no previously issued receipt is invalidated and TESTED keeps its meaning.",
6
6
  "type": "object",
7
7
  "additionalProperties": false,
8
8
  "required": ["schema_version", "skill", "suite", "run", "results", "comparison", "verification_level", "receipt_hash"],
9
+ "allOf": [
10
+ {
11
+ "description": "TESTED tightening — the interop relaxations (null hashes, external surface, null comparison, transcripts 'none') are ONLY available to receipts below TESTED. A TESTED receipt must carry the full evidence chain, exactly as before the interop revision.",
12
+ "if": { "required": ["verification_level"], "properties": { "verification_level": { "const": "TESTED" } } },
13
+ "then": {
14
+ "properties": {
15
+ "skill": { "properties": { "content_hash": { "type": "string" } } },
16
+ "suite": {
17
+ "properties": {
18
+ "format": { "const": "agentskills.io/evals" },
19
+ "suite_hash": { "type": "string" }
20
+ }
21
+ },
22
+ "run": {
23
+ "properties": {
24
+ "surface": { "enum": ["api", "claude-cli", "openai-api", "openai-cli"] },
25
+ "transcripts": { "enum": ["retained-local", "hashes-only"] }
26
+ }
27
+ },
28
+ "comparison": {
29
+ "properties": {
30
+ "baseline_score": { "type": "number" },
31
+ "delta": { "type": "number" },
32
+ "delta_uncertainty": { "type": "number" }
33
+ }
34
+ },
35
+ "results": {
36
+ "properties": {
37
+ "cases": {
38
+ "items": {
39
+ "if": { "not": { "required": ["case_status"], "properties": { "case_status": { "const": "failed_timeout" } } } },
40
+ "then": {
41
+ "required": ["generation_hash", "judge_sample_hashes"],
42
+ "properties": {
43
+ "judge": { "properties": { "rubric_hash": { "type": "string" } } }
44
+ }
45
+ }
46
+ }
47
+ }
48
+ }
49
+ }
50
+ }
51
+ }
52
+ }
53
+ ],
9
54
  "properties": {
10
55
  "schema_version": {
11
56
  "type": "string",
12
- "const": "0.3"
57
+ "const": "0.3.1"
13
58
  },
14
59
  "skill": {
15
60
  "type": "object",
@@ -19,9 +64,14 @@
19
64
  "name": { "type": "string", "minLength": 1 },
20
65
  "version": { "type": "string", "minLength": 1 },
21
66
  "content_hash": {
22
- "type": "string",
23
- "description": "sha256 (hex) over SKILL.md + all bundled files in canonical path-sorted order.",
67
+ "type": ["string", "null"],
68
+ "description": "sha256 (hex) over SKILL.md + all bundled files in canonical path-sorted order. Null ONLY on an imported (DECLARED) receipt whose skill bytes were never seen; a TESTED receipt must carry the hash (see the TESTED tightening).",
24
69
  "pattern": "^[a-f0-9]{64}$"
70
+ },
71
+ "tokens": {
72
+ "type": "integer",
73
+ "minimum": 0,
74
+ "description": "v0.3.1 (optional). Estimated token size of the skill's SKILL.md (a coarse chars/4 proxy, not a model tokenizer), used for the value-per-token axis (delta per 1k skill tokens). See docs/methodology.html."
25
75
  }
26
76
  }
27
77
  },
@@ -30,10 +80,14 @@
30
80
  "additionalProperties": false,
31
81
  "required": ["format", "suite_hash", "case_count"],
32
82
  "properties": {
33
- "format": { "type": "string", "const": "agentskills.io/evals" },
34
- "suite_hash": {
83
+ "format": {
35
84
  "type": "string",
36
- "description": "sha256 (hex) over the canonicalized normalized case list.",
85
+ "minLength": 1,
86
+ "description": "Suite format. Driftproof-run (TESTED) receipts are always 'agentskills.io/evals' (see the TESTED tightening); an imported receipt names the source tool's format (e.g. 'skillgrade/eval.yaml')."
87
+ },
88
+ "suite_hash": {
89
+ "type": ["string", "null"],
90
+ "description": "sha256 (hex) over the canonicalized normalized case list. Null ONLY on an imported (DECLARED) receipt whose suite bytes were never seen.",
37
91
  "pattern": "^[a-f0-9]{64}$"
38
92
  },
39
93
  "case_count": { "type": "integer", "minimum": 0 }
@@ -42,7 +96,7 @@
42
96
  "run": {
43
97
  "type": "object",
44
98
  "additionalProperties": false,
45
- "required": ["model_id", "surface", "runner_version", "date_utc", "judge", "registry", "transcripts"],
99
+ "required": ["model_id", "provider", "surface", "runner_version", "date_utc", "judge", "registry", "transcripts"],
46
100
  "properties": {
47
101
  "model_id": { "type": "string", "minLength": 1 },
48
102
  "model_release_date": {
@@ -50,7 +104,35 @@
50
104
  "type": ["string", "null"],
51
105
  "pattern": "^\\d{4}-\\d{2}-\\d{2}$"
52
106
  },
53
- "surface": { "type": "string", "enum": ["api", "claude-cli"] },
107
+ "provider": {
108
+ "type": "string",
109
+ "description": "v0.3.1. The two-axis provider the target model ran on (registry `provider`, else inferred from the id).",
110
+ "enum": ["anthropic", "openai"]
111
+ },
112
+ "surface": {
113
+ "type": "string",
114
+ "enum": ["api", "claude-cli", "openai-api", "openai-cli", "external"],
115
+ "description": "'external' = the run happened on another tool's harness and was imported (never valid on a TESTED receipt)."
116
+ },
117
+ "source": {
118
+ "type": "string",
119
+ "minLength": 1,
120
+ "description": "Interop-additive (optional). Provenance of a converted receipt, e.g. 'imported/agent-skills-eval' or 'imported/skillgrade'. Absent on receipts Driftproof ran itself."
121
+ },
122
+ "surface_overhead_note": {
123
+ "type": "string",
124
+ "description": "v0.3.1 (optional). Present on the openai/cli surface: states the fixed Codex base-instruction preamble (~12–15k input tokens per call) that the harness prepends and does not control."
125
+ },
126
+ "status": {
127
+ "type": "string",
128
+ "description": "v0.3.1 (optional; default 'complete'). 'incomplete' when >=1 case persistently failed (e.g. failed_timeout) and was EXCLUDED from aggregates. A drift/durability report must not compute a verdict from an incomplete receipt.",
129
+ "enum": ["complete", "incomplete"]
130
+ },
131
+ "failed_case_count": {
132
+ "type": "integer",
133
+ "minimum": 0,
134
+ "description": "v0.3.1 (optional). Number of cases marked failed_timeout (excluded from aggregates)."
135
+ },
54
136
  "runner_version": { "type": "string", "minLength": 1 },
55
137
  "date_utc": {
56
138
  "type": "string",
@@ -64,8 +146,8 @@
64
146
  },
65
147
  "transcripts": {
66
148
  "type": "string",
67
- "description": "Transcript retention for this run. 'hashes-only' = only the sha256 hashes in results.cases are kept (the default). 'retained-local' = the raw generations + judge outputs were also written to transcripts/<receipt-id>/ (gitignored by default).",
68
- "enum": ["retained-local", "hashes-only"]
149
+ "description": "Transcript retention for this run. 'hashes-only' = only the sha256 hashes in results.cases are kept (the default). 'retained-local' = the raw generations + judge outputs were also written to transcripts/<receipt-id>/ (gitignored by default). 'none' = nothing retained, not even hashes — ONLY honest on an imported (DECLARED) receipt.",
150
+ "enum": ["retained-local", "hashes-only", "none"]
69
151
  },
70
152
  "judge": {
71
153
  "type": "object",
@@ -82,7 +164,7 @@
82
164
  "type": "string",
83
165
  "description": "How sampling params were controlled, e.g. 'api-temperature-0' or 'surface-controlled'."
84
166
  },
85
- "surface": { "type": "string", "enum": ["api", "claude-cli"] }
167
+ "surface": { "type": "string", "enum": ["api", "claude-cli", "openai-api", "openai-cli", "external"] }
86
168
  }
87
169
  }
88
170
  }
@@ -97,10 +179,23 @@
97
179
  "items": {
98
180
  "type": "object",
99
181
  "additionalProperties": false,
100
- "required": ["id", "mode", "outcome", "score", "mean", "stddev", "samples", "judge", "generation_hash", "judge_sample_hashes"],
182
+ "required": ["id", "mode"],
183
+ "allOf": [
184
+ {
185
+ "description": "A completed case carries the full sampled band + hashes; a failed_timeout case is recorded WITHOUT fabricated samples (it is excluded from aggregates).",
186
+ "if": { "required": ["case_status"], "properties": { "case_status": { "const": "failed_timeout" } } },
187
+ "then": { "required": ["id", "mode", "case_status"] },
188
+ "else": { "required": ["outcome", "score", "mean", "stddev", "samples", "judge"] }
189
+ }
190
+ ],
101
191
  "properties": {
102
192
  "id": { "type": "string", "minLength": 1 },
103
193
  "mode": { "type": "string", "enum": ["with_skill", "baseline"] },
194
+ "case_status": {
195
+ "type": "string",
196
+ "description": "v0.3.1 (optional; default 'ok'). 'failed_timeout' = the case's model/judge call persistently timed out after retries; the case is recorded but EXCLUDED from aggregates/verdicts — no samples/hashes are fabricated for it.",
197
+ "enum": ["ok", "failed_timeout"]
198
+ },
104
199
  "outcome": {
105
200
  "type": "string",
106
201
  "description": "borderline = the threshold lies within mean +/- stddev.",
@@ -127,13 +222,31 @@
127
222
  },
128
223
  "threshold": { "type": ["number", "null"], "minimum": 0, "maximum": 1 },
129
224
  "reason": { "type": "string" },
225
+ "checks": {
226
+ "type": "array",
227
+ "description": "v0.3.1 (optional). Deterministic post-check results for this (case, mode): structural/regex assertions run on the model output ALONGSIDE the judge. Supplementary evidence reported as a separate column — NOT folded into the outcome/band verdict.",
228
+ "items": {
229
+ "type": "object",
230
+ "additionalProperties": false,
231
+ "required": ["name", "kind", "pass"],
232
+ "properties": {
233
+ "name": { "type": "string", "minLength": 1 },
234
+ "kind": { "type": "string", "enum": ["regex", "contains", "not_contains", "min_length"] },
235
+ "pass": { "type": "boolean" }
236
+ }
237
+ }
238
+ },
130
239
  "judge": {
131
240
  "type": "object",
132
241
  "additionalProperties": false,
133
242
  "required": ["model_id", "rubric_hash"],
134
243
  "properties": {
135
244
  "model_id": { "type": "string", "minLength": 1 },
136
- "rubric_hash": { "type": "string", "pattern": "^[a-f0-9]{64}$" }
245
+ "rubric_hash": {
246
+ "type": ["string", "null"],
247
+ "description": "Null ONLY on an imported (DECLARED) receipt whose rubric bytes were never seen.",
248
+ "pattern": "^[a-f0-9]{64}$"
249
+ }
137
250
  }
138
251
  }
139
252
  }
@@ -156,12 +269,22 @@
156
269
  "required": ["with_skill_score", "baseline_score", "delta", "delta_uncertainty"],
157
270
  "properties": {
158
271
  "with_skill_score": { "type": "number", "minimum": 0, "maximum": 1 },
159
- "baseline_score": { "type": "number", "minimum": 0, "maximum": 1 },
160
- "delta": { "type": "number", "minimum": -1, "maximum": 1 },
272
+ "baseline_score": {
273
+ "type": ["number", "null"],
274
+ "minimum": 0,
275
+ "maximum": 1,
276
+ "description": "Null ONLY on an imported (DECLARED) receipt from a tool with no baseline mode — never a fabricated 0."
277
+ },
278
+ "delta": {
279
+ "type": ["number", "null"],
280
+ "minimum": -1,
281
+ "maximum": 1,
282
+ "description": "Null when baseline_score is null (no baseline mode was run)."
283
+ },
161
284
  "delta_uncertainty": {
162
- "type": "number",
285
+ "type": ["number", "null"],
163
286
  "minimum": 0,
164
- "description": "Combined uncertainty of the delta (quadrature sum of the two aggregate bands)."
287
+ "description": "Combined uncertainty of the delta (quadrature sum of the two aggregate bands). Null when delta is null."
165
288
  }
166
289
  }
167
290
  },