driftproof 0.11.2 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/counts.js ADDED
@@ -0,0 +1,104 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Receipt counts (spec 043, receipt spec v0.8). Two counts describe how a receipt's data
5
+ // came to be, and they are kept apart because they measure different things (Report 006:
6
+ // generation spread ran 3 to 7 times judge spread on the same cases):
7
+ //
8
+ // generations_per_arm independent candidate generations
9
+ // judge_samples_per_generation judge samples taken of each generation
10
+ //
11
+ // A count sits at the narrowest scope that is honest about it: run.counts when it is
12
+ // the same for every case of every arm generated in this run, run.arms.<mode>.counts
13
+ // when it holds for one arm, results.cases[i].counts otherwise. An absent count is
14
+ // unknown, never 1 (agentskills discussion #544, comment 18409751, rule 1).
15
+
16
+ const KINDS = ['generations_per_arm', 'judge_samples_per_generation'];
17
+
18
+ // An arm that carries its own generated_at was generated in an earlier run. run.counts
19
+ // describes this run's generations, so an archived arm never inherits it.
20
+ function archived(receipt, mode) {
21
+ const a = receipt.run && receipt.run.arms && receipt.run.arms[mode];
22
+ return !!(a && a.generated_at);
23
+ }
24
+
25
+ const has = (o, k) => !!o && Object.prototype.hasOwnProperty.call(o, k) && Number.isInteger(o[k]);
26
+
27
+ // Where a reader looks, narrowest first.
28
+ const ORDER = ['case', 'arm', 'run'];
29
+ const SCOPES = {
30
+ case: (r, c) => c.counts,
31
+ arm: (r, c) => { const a = r.run && r.run.arms && r.run.arms[c.mode]; return a && a.counts; },
32
+ run: (r) => r.run && r.run.counts,
33
+ };
34
+
35
+ // The count of `kind` for results.cases[index], or null when the receipt does not
36
+ // establish it. run.judge.samples is the run-scope judge count (R-5), read only where
37
+ // run.counts is silent and only for an arm generated in this run.
38
+ function resolveCount(receipt, index, kind) {
39
+ if (!KINDS.includes(kind)) throw new Error(`unknown count kind: ${kind}`);
40
+ const c = receipt.results.cases[index];
41
+ for (const scope of ORDER) {
42
+ const at = SCOPES[scope](receipt, c);
43
+ if (has(at, kind)) return at[kind];
44
+ if (scope === 'run' && kind === 'judge_samples_per_generation' && receipt.run && receipt.run.judge && Number.isInteger(receipt.run.judge.samples)) return receipt.run.judge.samples;
45
+ if (scope === 'arm' && archived(receipt, c.mode)) return null;
46
+ }
47
+ return null;
48
+ }
49
+
50
+ // Which model generated an arm, and when: the arm's override, else the run's.
51
+ function resolveArm(receipt, mode) {
52
+ const a = (receipt.run && receipt.run.arms && receipt.run.arms[mode]) || {};
53
+ return {
54
+ model_id: a.model_id || receipt.run.model_id,
55
+ generated_at: a.generated_at || receipt.run.generated_at || null,
56
+ archived: archived(receipt, mode),
57
+ };
58
+ }
59
+
60
+ // Write counts at the narrowest honest scope (spec 043 AC-3). `perCase` is one entry per
61
+ // results.cases index: { index, counts: { <kind>: integer } } with a kind left out when
62
+ // the writer cannot establish it. Mutates and returns the receipt; call before sealing.
63
+ function placeCounts(receipt, perCase) {
64
+ const cases = receipt.results.cases;
65
+ const fresh = perCase.filter((p) => !archived(receipt, cases[p.index].mode));
66
+ const put = (obj, kind, v) => { obj.counts = { ...(obj.counts || {}), [kind]: v }; };
67
+ for (const kind of KINDS) {
68
+ const entries = fresh.map((p) => ({ p, v: p.counts && Number.isInteger(p.counts[kind]) ? p.counts[kind] : null }));
69
+ if (!entries.length) continue;
70
+ const allKnown = entries.length === cases.filter((c) => !archived(receipt, c.mode)).length && entries.every((e) => e.v !== null);
71
+ const values = new Set(entries.map((e) => e.v));
72
+ if (allKnown && values.size === 1) { receipt.run.counts = { ...(receipt.run.counts || {}), [kind]: entries[0].v }; continue; }
73
+ const byMode = new Map();
74
+ for (const e of entries) { const m = cases[e.p.index].mode; if (!byMode.has(m)) byMode.set(m, []); byMode.get(m).push(e); }
75
+ for (const [mode, es] of byMode) {
76
+ const modeKnown = allKnown && es.length === cases.filter((c) => c.mode === mode).length;
77
+ if (modeKnown && new Set(es.map((e) => e.v)).size === 1) {
78
+ receipt.run.arms = receipt.run.arms || {};
79
+ receipt.run.arms[mode] = receipt.run.arms[mode] || {};
80
+ put(receipt.run.arms[mode], kind, es[0].v);
81
+ } else {
82
+ for (const e of es) if (e.v !== null) put(cases[e.p.index], kind, e.v);
83
+ }
84
+ }
85
+ }
86
+ return receipt;
87
+ }
88
+
89
+ // The judge-sample count, or null when the receipt does not establish one at run scope.
90
+ function judgeSamplesOrNull(receipt) {
91
+ const j = receipt && receipt.run && receipt.run.judge;
92
+ if (j && Number.isInteger(j.samples)) return j.samples;
93
+ const c = receipt && receipt.run && receipt.run.counts;
94
+ return has(c, 'judge_samples_per_generation') ? c.judge_samples_per_generation : null;
95
+ }
96
+
97
+ // v0.8 loader rule (spec 043 AC-2): the two spellings of the run's judge count agree.
98
+ function judgeCountsDisagree(receipt) {
99
+ const j = receipt && receipt.run && receipt.run.judge;
100
+ const c = receipt && receipt.run && receipt.run.counts;
101
+ return !!(j && Number.isInteger(j.samples) && has(c, 'judge_samples_per_generation') && c.judge_samples_per_generation !== j.samples);
102
+ }
103
+
104
+ module.exports = { KINDS, ORDER, resolveCount, resolveArm, placeCounts, judgeSamplesOrNull, judgeCountsDisagree };
package/lib/decision.js CHANGED
@@ -17,10 +17,10 @@
17
17
  // here, in lib/, rather than in the shell, because `bin/driftproof badge` needs
18
18
  // it too and because a decision state is a property of a receipt.
19
19
 
20
- const fs = require('fs');
21
20
  const path = require('path');
22
21
  const { EFFECT_FLOOR } = require('../config');
23
- const { shortModel, githubOutputEntry, receiptVerdict, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
22
+ const { shortModel, githubOutputEntry, receiptVerdict, readCases, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
23
+ const { duplicateCaseRows, ambiguityLine, suiteNarrowed, listReceipts, fileSlug } = require('./receipt');
24
24
 
25
25
  // The six decision states (spec 030 AC-4).
26
26
  //
@@ -86,40 +86,57 @@ function failsJob(state, { failOnRegression = true } = {}) {
86
86
  // the fail-safe direction.
87
87
  function decisionState(receipt) {
88
88
  if (!receipt) return 'refused';
89
+ // Spec 050 AC-3: a receipt with two rows for one (id, mode) is `refused` before
90
+ // anything else is read. It is not `not measured`, which does not fail a job: the
91
+ // receipt may well have measured a regression, and which row a reader keeps decides
92
+ // whether that regression is seen. `refused` fails closed whatever
93
+ // fail-on-regression says, which is the direction an ambiguity has to fail.
94
+ if (duplicateCaseRows(receipt).length) return 'refused';
89
95
  const level = receipt.verification_level;
90
96
  const kind = receipt.run && receipt.run.answered_by ? receipt.run.answered_by.kind : undefined;
91
97
  if (level !== 'TESTED' || kind !== 'model') return 'not measured';
98
+ // Spec 062 (register row 1): THE MEASURED CASES BEFORE THE INCOMPLETE RULE. Until this an
99
+ // incomplete run read `inconclusive` before any case was read, so one unmeasured case hid a
100
+ // regression another case had measured: the same receipt read `regression`, exit 1, with every
101
+ // case measured, and `inconclusive`, exit 0, with one with-skill arm unmeasured (the register's
102
+ // pass 2 probe 7). A case that separated down was measured whatever happened to the others.
103
+ if (readCases(receipt).some((c) => c.state === 'separated-down')) return 'regression';
92
104
  const cmp = receipt.comparison || {};
93
- if ((receipt.run && receipt.run.status === 'incomplete') || typeof cmp.delta !== 'number') return 'inconclusive';
105
+ // Incomplete: a case failed (run.status), or the receipt ran fewer cases than its suite holds
106
+ // (spec 062, register row 4).
107
+ if ((receipt.run && receipt.run.status === 'incomplete') || suiteNarrowed(receipt) || typeof cmp.delta !== 'number') return 'inconclusive';
94
108
  // Spec 035: the measured states are the receipt verdict's, read per case by the
95
109
  // band rule (lib/verdict.js receiptVerdict), not the aggregate delta against the floor.
96
- // receiptVerdict refuses on the same four conditions the two clauses above split
97
- // between `not measured` and `inconclusive`, so it cannot answer NOT_MEASURED here.
98
- return MEASURED_STATE[receiptVerdict(receipt).verdict];
110
+ const verdict = receiptVerdict(receipt).verdict;
111
+ // Spec 062: every verdict has a state. NOT_MEASURED is reachable here - a receipt whose every
112
+ // case was dropped reads it by spec 036's no_readable_case rung - and until this it had no row
113
+ // in the table, so the state was undefined and the model left the set (pass 6 V3).
114
+ if (!Object.hasOwn(MEASURED_STATE, verdict)) throw new Error(`decisionState: the verdict ${JSON.stringify(verdict)} has no decision state`);
115
+ return MEASURED_STATE[verdict];
99
116
  }
100
117
 
101
- const MEASURED_STATE = { REGRESSED: 'regression', PASSED: 'helped', UNDERPOWERED: 'underpowered', NO_EFFECT: 'no detected effect' };
118
+ const MEASURED_STATE = { REGRESSED: 'regression', PASSED: 'helped', UNDERPOWERED: 'underpowered', NO_EFFECT: 'no detected effect', NOT_MEASURED: 'not measured' };
102
119
 
103
120
  // The worst state in a set, by STATE_ORDER. An empty set has no decision, and
104
- // says so with null rather than defaulting to something benign.
121
+ // says so with null rather than defaulting to something benign. A state this
122
+ // ordering does not know is refused (spec 062): skipping it dropped its model
123
+ // from the decision, which is a pass by omission.
105
124
  function worstState(states) {
106
125
  let worst = null;
107
126
  for (const s of states) {
108
127
  const i = STATE_ORDER.indexOf(s);
109
- if (i < 0) continue;
128
+ if (i < 0) throw new Error(`worstState: unknown decision state ${JSON.stringify(s)}`);
110
129
  if (worst === null || i < STATE_ORDER.indexOf(worst)) worst = s;
111
130
  }
112
131
  return worst;
113
132
  }
114
133
 
115
- // Every receipt in a directory, excluding the summaries and the badge writer's
116
- // own earlier output. Mirrors what action/run.sh's glob selected from, so the
117
- // set this reads is the set that was there to be read.
134
+ // Every receipt file in a directory, by lib/receipt.js listReceipts (spec 062): the files that
135
+ // parse and carry receipt_hash or results and are no sidecar by name and shape (spec 069), and the
136
+ // files that do not parse. Until spec 062 it was
137
+ // every `.json` but the summaries and badge.json, so a regrade sidecar was read as a receipt.
118
138
  function receiptFiles(dir) {
119
- return fs.readdirSync(dir)
120
- .filter((f) => f.endsWith('.json') && !f.includes('.summary.') && f !== 'badge.json')
121
- .sort()
122
- .map((f) => path.join(dir, f));
139
+ return listReceipts(dir).map((e) => e.file);
123
140
  }
124
141
 
125
142
  // Parse the `models` input the same way the runner does: comma-separated, with
@@ -152,9 +169,11 @@ function matchesModel(receiptModelId, requested) {
152
169
  function fileNamesModel(file, requested) {
153
170
  const base = path.basename(file);
154
171
  // Both spellings, for the reason `matchesModel` takes both: a receipt for
155
- // `claude-haiku-4-5-20251001` answers a request for `claude-haiku-4-5`.
172
+ // `claude-haiku-4-5-20251001` answers a request for `claude-haiku-4-5`. Each is
173
+ // looked for in the form the runner writes it, lib/receipt.js fileSlug (spec 062): an
174
+ // id with a dot or a capital is not in the file name as itself.
156
175
  for (const id of new Set([requested, shortModel(requested)])) {
157
- if (base.includes(`-${id}-`)) return true;
176
+ if (base.includes(`-${fileSlug(id)}-`)) return true;
158
177
  }
159
178
  return false;
160
179
  }
@@ -166,37 +185,49 @@ function fileNamesModel(file, requested) {
166
185
  // row that is simply absent. That is the whole of AC-2: absence is not a pass.
167
186
  function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
168
187
  const requested = parseModels(requestedCsv);
169
- const files = receiptFiles(dir);
170
- const loaded = files.map((f) => {
171
- let receipt = null, error = null;
172
- try { receipt = JSON.parse(fs.readFileSync(f, 'utf8')); } catch (e) { error = e.message; }
173
- return { file: f, receipt, error };
174
- });
188
+ const loaded = listReceipts(dir);
189
+ const files = loaded.map((l) => l.file);
175
190
  const used = new Set();
176
191
  const rows = requested.map((model) => {
177
- const hit = loaded.find((l) => !used.has(l.file) && l.receipt
192
+ // Spec 050 AC-6: EVERY readable receipt that matches, not the first. Until this the
193
+ // row took `loaded.find(...)`, the first match in sorted filename order, and reported
194
+ // the rest as unexpected. A job with two Driftproof steps shared one directory, so
195
+ // the second step's model row was filled by whichever skill's receipt sorted first
196
+ // (an outside correctness audit, finding 2; spec 030's F-1). Two receipts for one
197
+ // model are a question nobody answered, so the row is `refused`, names every file,
198
+ // and reads none of them.
199
+ const matches = loaded.filter((l) => !used.has(l.file) && l.receipt
178
200
  && l.receipt.run && matchesModel(l.receipt.run.model_id, model));
179
- if (hit) used.add(hit.file);
201
+ for (const m of matches) used.add(m.file);
202
+ const many = matches.length > 1;
203
+ const hit = matches.length === 1 ? matches[0] : null;
180
204
  // A receipt that exists and cannot be parsed is UNREADABLE, not absent.
181
205
  // Both fail closed, and the row says which - the register's
182
206
  // absence-vs-unreadable distinction, kept at the point it is decided.
183
- const unreadable = !hit && loaded.find((l) => !used.has(l.file) && l.error
207
+ const unreadable = !hit && !many && loaded.find((l) => !used.has(l.file) && l.error
184
208
  && fileNamesModel(l.file, model));
185
209
  if (unreadable) used.add(unreadable.file);
186
210
  const receipt = hit ? hit.receipt : null;
211
+ // Spec 050 AC-3: one receipt, but one with two rows for a case and arm.
212
+ const dups = receipt ? duplicateCaseRows(receipt) : [];
213
+ const ambiguous = many
214
+ ? `${matches.length} receipts for this model (${matches.map((m) => path.basename(m.file)).join(', ')}); none was read`
215
+ : (dups.length ? `receipt ambiguous: ${ambiguityLine(dups)}` : null);
187
216
  const state = decisionState(receipt);
188
217
  return {
189
218
  model,
190
219
  state,
191
- delta: receipt && receipt.comparison && typeof receipt.comparison.delta === 'number'
220
+ // An ambiguous row carries no delta: a delta is a reading, and none was taken.
221
+ delta: !ambiguous && receipt && receipt.comparison && typeof receipt.comparison.delta === 'number'
192
222
  ? receipt.comparison.delta : null,
193
- file: hit ? path.basename(hit.file) : (unreadable ? path.basename(unreadable.file) : null),
223
+ file: many ? null : hit ? path.basename(hit.file) : (unreadable ? path.basename(unreadable.file) : null),
194
224
  // Spec 035 AC-3: what an underpowered row would have needed, from the same reading.
195
225
  drawsNeeded: state === 'underpowered' ? receiptVerdict(receipt).drawsNeeded : null,
196
226
  // The distinction is carried on the ROW, not left in a sentence, because
197
227
  // the enforcement message has to say which of the two happened (AC-2's
198
228
  // mutation class, absence-vs-unreadable).
199
229
  unreadable: unreadable ? unreadable.error : null,
230
+ ambiguous,
200
231
  // A SHORT CAUSE, not the parser's text. V8's JSON.parse message echoes the
201
232
  // first bytes of the input - "Unexpected token '|', \"|{bad\" is not valid
202
233
  // JSON" - and this string is rendered into a markdown table cell, where one
@@ -204,7 +235,7 @@ function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
204
235
  // reaches the annotation, which is not a table; the cell carries the cause,
205
236
  // and the filename is already beside it in the same cell (F-2 of
206
237
  // 2026-09-12).
207
- reason: unreadable ? 'receipt unreadable (not valid JSON)' : (hit ? null : 'no receipt for this model'),
238
+ reason: ambiguous || (unreadable ? 'receipt unreadable (not valid JSON)' : (hit ? null : 'no receipt for this model')),
208
239
  };
209
240
  });
210
241
  // Receipts the run wrote for models nobody requested are reported rather than
@@ -226,9 +257,14 @@ function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
226
257
  // split it by CAUSE: nothing was written for this model, or something was
227
258
  // written and cannot be read. Both fail closed; they are not the same fact,
228
259
  // and the failure has to say which.
229
- absent: rows.filter((r) => r.state === 'refused' && !r.unreadable).map((r) => r.model),
260
+ absent: rows.filter((r) => r.state === 'refused' && !r.unreadable && !r.ambiguous).map((r) => r.model),
230
261
  unreadable: rows.filter((r) => r.state === 'refused' && r.unreadable)
231
262
  .map((r) => ({ model: r.model, file: r.file, error: r.unreadable })),
263
+ // Spec 050: a third cause of `refused`, apart from the other two for the same reason
264
+ // they are apart from each other. Something was written, and it can be read, but it
265
+ // does not say one thing.
266
+ ambiguous: rows.filter((r) => r.state === 'refused' && r.ambiguous)
267
+ .map((r) => ({ model: r.model, reason: r.ambiguous })),
232
268
  unexpected,
233
269
  receiptCount: files.length,
234
270
  requestedCount: requested.length,
@@ -390,6 +426,11 @@ function enforcementLines(d, { failOnRegression = true } = {}) {
390
426
  + `${d.unreadable.map((u) => `${u.model} exists and is unreadable (${u.file}: ${oneLine(u.error)})`).join('; ')}. `
391
427
  + `Failing closed: a receipt that cannot be read is not a pass, and is not the same as one that was never written.`);
392
428
  }
429
+ if (d.ambiguous && d.ambiguous.length) {
430
+ L.push(`::error title=Driftproof::The receipt set is ambiguous for `
431
+ + `${d.ambiguous.map((a) => `${a.model}: ${oneLine(a.reason, 240)}`).join('; ')}. `
432
+ + `Failing closed: a decision that depends on which receipt or row was read is not a pass.`);
433
+ }
393
434
  if (d.regressed.length) {
394
435
  const deltas = d.rows.filter((r) => r.state === 'regression')
395
436
  .map((r) => `${r.model} (delta ${r.delta})`).join(', ');
package/lib/diff.js CHANGED
@@ -1,6 +1,7 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
+ const { judgeSamplesOrNull } = require('./counts');
4
5
  const { bandVerdict, round, WITHIN_NOISE } = require('./stats');
5
6
  const { EFFECT_FLOOR } = require('../config');
6
7
  const { revisionHeadline } = require('./revision');
@@ -158,6 +159,12 @@ function revisionPairProblem(a, b) {
158
159
  return null;
159
160
  }
160
161
 
162
+ // The run's judge-sample count as a reader shows it: the number, or 'unknown'.
163
+ function judgeN(r) {
164
+ const n = judgeSamplesOrNull(r);
165
+ return n === null ? 'unknown' : n;
166
+ }
167
+
161
168
  function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' } = {}) {
162
169
  const revision = mode === 'revision';
163
170
 
@@ -253,7 +260,7 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
253
260
  L.push(`| surface (held) | ${a.run.surface} | ${b.run.surface} |`);
254
261
  L.push(`| suite_hash (held) | \`${short(a.suite.suite_hash)}\` | \`${short(b.suite.suite_hash)}\` |`);
255
262
  L.push(`| run date (UTC) | ${a.run.date_utc} | ${b.run.date_utc} |`);
256
- L.push(`| judge samples/case | ${(a.run.judge || {}).samples || 1} | ${(b.run.judge || {}).samples || 1} |`);
263
+ L.push(`| judge samples/case | ${judgeN(a)} | ${judgeN(b)} |`);
257
264
  L.push(`| with_skill (mean ± band) | ${bandStr(aAgg)} | ${bandStr(bAgg)} |`);
258
265
  L.push(`| baseline score | ${pct(a.comparison.baseline_score)} | ${pct(b.comparison.baseline_score)} |`);
259
266
  L.push(`| skill lift (Δ) | ${fmt(a.comparison.delta)} | ${fmt(b.comparison.delta)} |`);
@@ -268,7 +275,7 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
268
275
  L.push(`| model | \`${a.run.model_id}\` | \`${b.run.model_id}\` |`);
269
276
  L.push(`| run date (UTC) | ${a.run.date_utc} | ${b.run.date_utc} |`);
270
277
  L.push(`| surface | ${a.run.surface} | ${b.run.surface} |`);
271
- L.push(`| judge samples/case | ${(a.run.judge || {}).samples || 1} | ${(b.run.judge || {}).samples || 1} |`);
278
+ L.push(`| judge samples/case | ${judgeN(a)} | ${judgeN(b)} |`);
272
279
  L.push(`| skill content_hash | \`${short(a.skill.content_hash)}\` | \`${short(b.skill.content_hash)}\` |`);
273
280
  L.push(`| suite_hash | \`${short(a.suite.suite_hash)}\` | \`${short(b.suite.suite_hash)}\` |`);
274
281
  L.push(`| with_skill (mean ± band) | ${bandStr(aAgg)} | ${bandStr(bAgg)} |`);
@@ -289,7 +296,9 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
289
296
  if (!revision && a.skill.content_hash !== b.skill.content_hash) warnings.push('skill content_hash differs — the skill itself changed between receipts, so drift mixes skill edits with model drift.');
290
297
  if (!revision && a.suite.suite_hash !== b.suite.suite_hash) warnings.push('suite_hash differs — the eval suite changed; per-case comparison may be misleading.');
291
298
  if (a.skill.name !== b.skill.name) warnings.push(`different skills (${a.skill.name} vs ${b.skill.name}) — comparison is not meaningful.`);
292
- if ((a.run.judge || {}).samples <= 1 || (b.run.judge || {}).samples <= 1) warnings.push('one or both receipts are single-sample (no bands) — non-overlap can only be trusted when both sides are sampled.');
299
+ // v0.8 (spec 043 AC-1): an absent judge count is unknown, said as such, never read as 1.
300
+ if (judgeN(a) === 'unknown' || judgeN(b) === 'unknown') warnings.push('the judge-sample count is unknown on one or both receipts: the receipt does not establish it, so whether the bands are sampled cannot be read from it.');
301
+ if (judgeN(a) === 1 || judgeN(b) === 1) warnings.push('one or both receipts are single-sample (no bands): non-overlap can only be trusted when both sides are sampled.');
293
302
  // WHAT SUPPRESSED THE VERDICT IS SAID, AND IT IS SAID CORRECTLY (spec 016 AC-3
294
303
  // and AC-4, closing F-015-B).
295
304
  //
package/lib/export.js CHANGED
@@ -8,13 +8,16 @@
8
8
  // Documented in docs/interop.md; snapshot-tested in the gate.
9
9
 
10
10
  const { verdictFromReceipt } = require('./verdict');
11
+ const { SUMMARY_SIDECAR } = require('./receipt');
11
12
 
12
- const SUMMARY_FORMAT = 'driftproof/summary';
13
+ const SUMMARY_FORMAT = SUMMARY_SIDECAR.format;
13
14
  const SUMMARY_FORMAT_VERSION = '1';
14
15
 
15
16
  // Build the summary object for one receipt. Deterministic for a given receipt:
16
17
  // fixed key order, no export-time timestamps. `reportUrl` is caller-supplied
17
18
  // (receipts do not know where their report lives), else null.
19
+ const { judgeSamplesOrNull } = require('./counts');
20
+
18
21
  function toSummaryJson(receipt, { reportUrl = null } = {}) {
19
22
  const agg = receipt.results.aggregates;
20
23
  const cmp = receipt.comparison || {};
@@ -41,7 +44,8 @@ function toSummaryJson(receipt, { reportUrl = null } = {}) {
41
44
  source: receipt.run.source || 'driftproof',
42
45
  judge: {
43
46
  model_id: (receipt.results.cases.find((c) => c.judge) || { judge: { model_id: 'unknown' } }).judge.model_id,
44
- samples: (receipt.run.judge || {}).samples || 1,
47
+ // v0.8 (spec 043 AC-1): null when the receipt does not establish the count, never 1.
48
+ samples: judgeSamplesOrNull(receipt),
45
49
  },
46
50
  receipt_hash: receipt.receipt_hash,
47
51
  report_url: reportUrl,