driftproof 0.8.1 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/judge.js CHANGED
@@ -58,10 +58,31 @@ function rubricHash(rubric) {
58
58
  return sha256(JUDGE_SYSTEM + '\n---\n' + String(rubric || '').trim());
59
59
  }
60
60
 
61
- function clamp01(n) {
62
- const x = Number(n);
63
- if (!Number.isFinite(x)) return 0;
64
- return Math.max(0, Math.min(1, x));
61
+ // v0.6 (spec 026 AC-11): a digest over the grading TEMPLATE with its three
62
+ // slots empty, recorded once per run as run.judge.prompt_template_hash. The
63
+ // rubric_hash above binds a grade to the case's rubric; this binds every grade
64
+ // of the run to the words around it. A different template is a different judge,
65
+ // and `diff` computes no verdict across one.
66
+ function promptTemplateHash() {
67
+ return sha256(JUDGE_SYSTEM + '\n---\n' + buildJudgePrompt({ task: '', response: '', rubric: '' }));
68
+ }
69
+
70
+ // The score a judge reply carries, or null when it carries none (spec 026
71
+ // AC-3, F2). A non-numeric or non-finite value is not a score; a number outside
72
+ // [0, 1] is on a scale the rubric did not ask for and is not clamped into one
73
+ // (85 is not 1.0). This replaced clamp01, whose silent 0 for a non-finite value
74
+ // scored an absent output as the worst possible one.
75
+ // Stop reasons that mean the surface cut the reply at its output cap: the
76
+ // Messages API's and the claude CLI's `max_tokens`, Chat Completions'
77
+ // `length`. A cut reply is a partial answer (spec 026 AC-8, F4).
78
+ const TRUNCATION_STOP_REASONS = new Set(['max_tokens', 'length']);
79
+ function isTruncated(stopReason) { return TRUNCATION_STOP_REASONS.has(String(stopReason || '')); }
80
+
81
+ function scoreOf(parsed) {
82
+ const x = Number(parsed && parsed.score);
83
+ if (!Number.isFinite(x)) return null;
84
+ if (x < 0 || x > 1) return null;
85
+ return x;
65
86
  }
66
87
 
67
88
  // Judge settings for the JUDGE model's surface. Determinism where the surface
@@ -78,25 +99,50 @@ function judgeSettings(samples, judgeModel) {
78
99
  return { samples, temperature: null, sampling: 'surface-controlled', surface };
79
100
  }
80
101
 
81
- // Grade one response once. Returns { score, reason, raw }.
102
+ // Grade one response once. Returns { score, reason, raw, ... } for a measured
103
+ // sample, or { unmeasured: true, reason, raw, ... } when the reply carries no
104
+ // score: empty, unparseable, no numeric score, or a score outside [0, 1]. A
105
+ // zero asserts a measurement ("the response satisfied none of the rubric");
106
+ // each of these is the absence of one, and no sample enters any statistic
107
+ // (spec 026 AC-3). The 2026-07 rule that an unparseable judge must never
108
+ // silently pass is kept by the stronger rule: it never silently scores at all.
82
109
  // `raw` is the judge's verbatim output text (hashed into the receipt for
83
110
  // transcript auditability, and optionally retained under --keep-transcripts).
84
111
  async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature, trusted = false }) {
85
112
  const prompt = buildJudgePrompt({ task, response, rubric });
86
- const { text, attempts, usage } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature, trusted });
113
+ const out = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature, trusted });
114
+ const { text, attempts, usage } = out;
115
+ const base = { raw: String(text || ''), attempts: attempts || 1, usage, reply: out };
116
+ // A judge reply cut at its output cap carries no complete grade.
117
+ if (isTruncated(out.stopReason)) {
118
+ return { ...base, unmeasured: true, reason: `judge output truncated at the output cap (stop_reason ${out.stopReason})` };
119
+ }
120
+ if (!String(text || '').trim()) {
121
+ return { ...base, unmeasured: true, reason: 'judge output empty (the surface returned no text)' };
122
+ }
87
123
  let parsed;
88
124
  try {
89
125
  parsed = extractJsonObject(text);
90
126
  } catch (_e) {
91
- // Unsalvageable judge output → conservative 0 (a judge that can't be parsed
92
- // must never silently "pass"), tagged so the caller can see it happened.
93
- return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || ''), attempts: attempts || 1, usage };
127
+ return { ...base, unmeasured: true, reason: 'judge output unparseable' };
94
128
  }
95
- return { score: clamp01(parsed.score), reason: String(parsed.reason || '').slice(0, 300), raw: String(text || ''), attempts: attempts || 1, usage };
129
+ const score = scoreOf(parsed);
130
+ if (score === null) {
131
+ const x = parsed && parsed.score;
132
+ const reason = x === undefined || x === null ? 'judge output carries no numeric score'
133
+ : !Number.isFinite(Number(x)) ? `judge output carries no numeric score (score ${JSON.stringify(x)})`
134
+ : `judge score ${x} outside [0, 1]`;
135
+ return { ...base, unmeasured: true, reason };
136
+ }
137
+ return { ...base, score, reason: String(parsed.reason || '').slice(0, 300) };
96
138
  }
97
139
 
98
140
  // Grade a response N times and return the sampled distribution:
99
141
  // { samples:[scores], mean, stddev, reason, judge_settings, model_id, rubric_hash }
142
+ // or, when any sample is unmeasured, { unmeasured: true, reason, ... } with NO
143
+ // samples: a partial sample set must never become a band, and a draw one of
144
+ // whose judge samples carried no score is unmeasured as a whole (spec 026
145
+ // AC-3). The remaining samples are not taken.
100
146
  // `mean` ± `stddev` is the per-case confidence band (raw spread of the N scores)
101
147
  // used by the borderline-outcome rule and per-case drift band-overlap logic.
102
148
  // NO DEFAULT TIMEOUT HERE (spec 017 AC-2). This defaulted to 120000, which
@@ -112,6 +158,7 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
112
158
  const reasons = [];
113
159
  const rawTexts = [];
114
160
  const usages = [];
161
+ const replies = []; // v0.6: what answered each judge call, for the runner's attestation
115
162
  let attemptsTotal = 0;
116
163
  for (let i = 0; i < samples; i++) {
117
164
  let r;
@@ -126,9 +173,18 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
126
173
  }
127
174
  attemptsTotal += r.attempts || 1;
128
175
  usages.push(r.usage || null);
176
+ replies.push(r.reply || null);
177
+ rawTexts.push(r.raw || '');
178
+ if (r.unmeasured) {
179
+ return {
180
+ unmeasured: true, reason: r.reason, samples: [], mean: null, stddev: null,
181
+ sample_texts: rawTexts, sample_hashes: rawTexts.map((t) => sha256(t)),
182
+ judge_settings: settings, model_id: model, rubric_hash: rubricHash(rubric),
183
+ attempts: attemptsTotal, usage: sumUsage(usages), replies,
184
+ };
185
+ }
129
186
  scores.push(r.score);
130
187
  reasons.push(r.reason);
131
- rawTexts.push(r.raw || '');
132
188
  }
133
189
  return {
134
190
  samples: scores,
@@ -149,7 +205,8 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
149
205
  // EXCLUDED from every skill-value figure (lib/value.js): it is a cost we
150
206
  // impose to measure, not a cost of running the skill.
151
207
  usage: sumUsage(usages),
208
+ replies,
152
209
  };
153
210
  }
154
211
 
155
- module.exports = { gradeOnce, gradeSamples, judgeSettings, rubricHash, buildJudgePrompt, JUDGE_SYSTEM };
212
+ module.exports = { gradeOnce, gradeSamples, judgeSettings, rubricHash, buildJudgePrompt, promptTemplateHash, scoreOf, isTruncated, TRUNCATION_STOP_REASONS, JUDGE_SYSTEM };
package/lib/models.js CHANGED
@@ -66,6 +66,26 @@ function resolveRegistry(modelId) {
66
66
  return { id: canonical, entry, registered: !!entry };
67
67
  }
68
68
 
69
+ // Spec 026 AC-10 (F6): an id that is not a model does not run. After alias
70
+ // resolution the id must have the contract's model-id shape (letters, digits,
71
+ // . _ -) and resolve in the registry; otherwise the run is refused before any
72
+ // call, naming the id and the absolute path of the registry consulted
73
+ // (DRIFTPROOF_REGISTRY when set, else the packaged config/models.json).
74
+ // `registry: "unregistered"` stays a legal receipt value for IMPORTED receipts,
75
+ // whose model field is another tool's word; a run of ours never writes it.
76
+ const MODEL_ID_SHAPE = /^[A-Za-z0-9._-]+$/;
77
+ function assertRegistered(modelId, role = 'model') {
78
+ const given = String(modelId == null ? '' : modelId);
79
+ const { id, registered } = MODEL_ID_SHAPE.test(given) ? resolveRegistry(given) : { id: given, registered: false };
80
+ if (!MODEL_ID_SHAPE.test(given) || !registered) {
81
+ const why = !MODEL_ID_SHAPE.test(given) ? 'is not a model id (letters, digits, . _ - only)' : 'is not in the model registry';
82
+ const e = new Error(`${role} "${given}"${id !== given ? ` (resolved "${id}")` : ''} ${why}: ${REGISTRY_PATH}. An unregistered model does not run; add a registry row (see config/models.json) or point DRIFTPROOF_REGISTRY at a registry that carries it.`);
83
+ e.code = 'UNREGISTERED_MODEL'; e.model = given; e.registry = REGISTRY_PATH;
84
+ throw e;
85
+ }
86
+ return id;
87
+ }
88
+
69
89
  // The value stamped into receipt.run.registry.
70
90
  function registryStatus(modelId) {
71
91
  return resolveRegistry(modelId).registered ? 'registered' : 'unregistered';
@@ -140,23 +160,43 @@ function familyPredecessor(modelId) {
140
160
  return chooseFrom[0].id;
141
161
  }
142
162
 
143
- // Append a newly-discovered model id to the registry file (release trigger).
144
- // `firstSeen` is the first-seen date recorded as `released`. Idempotent: a no-op
145
- // if the id already exists. Returns the added entry (or null if already present).
146
- function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH } = {}) {
147
- const raw = JSON.parse(fs.readFileSync(registryPath, 'utf8'));
148
- if (raw.models.some((m) => m.id === id)) return null;
163
+ // THE PRICE OF A MODEL NOBODY HAS PRICED YET.
164
+ //
165
+ // READ THIS BEFORE TRUSTING A NUMBER IT RETURNS. Nothing here reads a price
166
+ // SOURCE. The family and the tier are inferred from the id, and the rate is
167
+ // looked up in the table below, so the result is a GUESS whose only input is the
168
+ // string. It is extracted into its own function, and named, because the guess
169
+ // was being made silently inside a function whose job looked like registration.
170
+ //
171
+ // KNOWN DEFECT, spec 021 F-W2, deliberately NOT fixed here. The Anthropic branch
172
+ // falls through to the OPUS rate of 5/25 for every family that is not `haiku` or
173
+ // `sonnet`. Fable's published rate is 10/50, exactly twice opus, which is the
174
+ // whole of the "the watcher read exactly half the snapshot on both fields"
175
+ // observation of 2026-09-03: it is not a halving and not a misread, it is a
176
+ // wrong-tier fallback. `DEFAULT_PRICE` above IS the conservative upper bound the
177
+ // comment on this function used to claim, and it is not consulted. Recorded on
178
+ // the spec 021 carry list, tagged 026, with the reason it is left alone.
179
+ //
180
+ // scripts/release-watch.js no longer registers anything, and records what this
181
+ // returns as `price_verified: false` beside a source naming this function.
182
+ function inferredPrice(id) {
149
183
  const family = familyOf(id);
150
184
  const provider = inferProviderFromId(id);
151
- // Best-effort tier from family; unknown families default to frontier pricing.
152
- // A newly-discovered id gets a CONSERVATIVE (upper-bound) price so the budget
153
- // guard never under-projects an unknown model — exact rates are set by hand
154
- // when the model is reviewed for a published run.
155
185
  const tier = /haiku|luna|mini|nano/.test(family) || /luna|mini|nano/.test(String(id)) ? 'cheap'
156
186
  : /sonnet|terra/.test(family) ? 'standard' : 'frontier';
157
187
  const price = provider === 'openai'
158
188
  ? (tier === 'cheap' ? { input: 1.0, output: 6.0 } : tier === 'standard' ? { input: 2.5, output: 15.0 } : { input: 5.0, output: 30.0 })
159
189
  : (family === 'haiku' ? { input: 1.0, output: 5.0 } : family === 'sonnet' ? { input: 3.0, output: 15.0 } : { input: 5.0, output: 25.0 });
190
+ return { family, provider, tier, price };
191
+ }
192
+
193
+ // Append a newly-discovered model id to the registry file (release trigger).
194
+ // `firstSeen` is the first-seen date recorded as `released`. Idempotent: a no-op
195
+ // if the id already exists. Returns the added entry (or null if already present).
196
+ function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH } = {}) {
197
+ const raw = JSON.parse(fs.readFileSync(registryPath, 'utf8'));
198
+ if (raw.models.some((m) => m.id === id)) return null;
199
+ const { family, provider, tier, price } = inferredPrice(id);
160
200
  const entry = {
161
201
  id, family, provider, released: firstSeen,
162
202
  input_price: price.input, output_price: price.output, tier,
@@ -171,5 +211,6 @@ function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH
171
211
  module.exports = {
172
212
  loadRegistry, resolveRegistry, registryStatus, priceForModel, isJudgeEligible,
173
213
  assertJudgeEligible, providerForModel, providerConfig, inferProviderFromId,
174
- familyOf, familyPredecessor, addDiscoveredModel, DEFAULT_PRICE, REGISTRY_PATH,
214
+ familyOf, familyPredecessor, addDiscoveredModel, inferredPrice, DEFAULT_PRICE, REGISTRY_PATH,
215
+ assertRegistered, MODEL_ID_SHAPE,
175
216
  };
package/lib/provider.js CHANGED
@@ -9,6 +9,7 @@ const path = require('path');
9
9
  const cp = require('child_process');
10
10
  const { withRetry, withTimeout } = require('./json');
11
11
  const { stubComplete, stubEnabled } = require('./stub');
12
+ const { readAnthropicApiResponse, readOpenaiApiResponse } = require('./usage');
12
13
  const {
13
14
  parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage, normalizeUsage,
14
15
  } = require('./usage');
@@ -134,12 +135,17 @@ const CODEX_OVERHEAD_NOTE =
134
135
  async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined, trusted = false }) {
135
136
  const surface = surfaceForModel(model);
136
137
  const provider = providerForSurface(surface);
137
- // Offline stub surface: canned completion, zero model calls. The receipt still
138
- // records the real surface/provider so a stub run is not mistaken for a genuine
139
- // one at read time — only the generation/judge TEXT is canned.
138
+ // Offline stub surface: canned completion, zero model calls. The reply says
139
+ // so: surface `stub`, answeredBy `stub`, no model reported, no stop reason,
140
+ // no isolation (nothing was spawned). Spec 026 F1: the receipt used to record
141
+ // the REAL surface name here and read TESTED while the text was canned; now
142
+ // the receipt's surface, level and answered_by are derived from what this
143
+ // function returned, and a stub run reads UNVERIFIED with surface stub. The
144
+ // requested model's provider is still reported, so a reader knows what was
145
+ // asked for.
140
146
  if (stubEnabled()) {
141
147
  const s = stubComplete({ system, prompt });
142
- return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface, provider, attempts: 1 };
148
+ return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface: 'stub', provider, attempts: 1, answeredBy: 'stub', reportedModels: null, stopReason: null, isolation: 'none' };
143
149
  }
144
150
 
145
151
  // Per-surface tolerance (see retryPolicyForSurface): a cli/subscription surface
@@ -177,7 +183,16 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
177
183
  try {
178
184
  const out = await withRetry(runner, { tries: policy.tries, baseDelayMs });
179
185
  const usage = normalizeUsage({ ...(out.usage || {}), wall_ms: wallMs });
180
- return { ...out, usage, wall_ms: wallMs, surface, provider, attempts };
186
+ // v0.6 (spec 026 AC-1, AC-2, AC-8): a model surface answered; what it said
187
+ // served the call, why the reply stopped, and which spawn path was taken
188
+ // are what the lane could read, null where its surface reports nothing.
189
+ return {
190
+ ...out, usage, wall_ms: wallMs, surface, provider, attempts,
191
+ answeredBy: 'model',
192
+ reportedModels: Array.isArray(out.reportedModels) && out.reportedModels.length ? out.reportedModels : null,
193
+ stopReason: typeof out.stopReason === 'string' && out.stopReason ? out.stopReason : null,
194
+ isolation: out.isolation || 'none',
195
+ };
181
196
  } catch (e) {
182
197
  // Surface the attempt count so a persistently-failing call can be charged for
183
198
  // (and, for a timeout, marked failed_timeout by the runner instead of fatal).
@@ -202,7 +217,8 @@ async function completeAnthropicApi({ system, prompt, model, maxTokens, temperat
202
217
  if (temperature !== undefined) params.temperature = temperature;
203
218
  const resp = await client.messages.create(params);
204
219
  const text = (resp.content || []).filter((b) => b.type === 'text').map((b) => b.text).join('');
205
- return { text, usage: parseAnthropicApiUsage(resp.usage) };
220
+ // v0.6: the response names the model that served it and why it stopped.
221
+ return { text, usage: parseAnthropicApiUsage(resp.usage), ...readAnthropicApiResponse(resp), isolation: 'none' };
206
222
  }
207
223
 
208
224
  // ── isolation: the eval-user hop (spec 022) ───────────────────────────────────
@@ -275,6 +291,10 @@ function isolatedEnv(user) {
275
291
  return { HOME: home, PATH: `${home}/.local/bin:/usr/bin:/bin` };
276
292
  }
277
293
 
294
+ // The isolation a plan records into the receipt (run.answered_by.isolation,
295
+ // spec 026 AC-2): the hop is `eval-user`, the legacy spawn `same-user`.
296
+ function isolationOf(plan) { return plan && plan.mode === 'isolated' ? 'eval-user' : 'same-user'; }
297
+
278
298
  // The spawn plan for one CLI call: { mode, user, file, args, env }. Pure, so the
279
299
  // gate can inspect what WOULD be spawned. Untrusted → the hop; trusted → the
280
300
  // legacy same-user spawn exactly as it was (env inherited; the claude lane still
@@ -376,9 +396,10 @@ async function completeClaudeCli({ system, prompt, model, timeoutMs, trusted = f
376
396
  throw new Error(hop || `claude CLI exited ${r.code}: ${r.stderr.slice(0, 400)}`);
377
397
  }
378
398
  const parsed = parseClaudeCliJson(r.stdout);
379
- if (!parsed) return { text: r.stdout.trim(), usage: null };
399
+ if (!parsed) return { text: r.stdout.trim(), usage: null, stopReason: null, reportedModels: null, isolation: isolationOf(plan) };
380
400
  if (parsed.isError) throw new Error(`claude CLI reported is_error: ${String(parsed.text || '').slice(0, 300)}`);
381
- return { text: String(parsed.text || '').trim(), usage: parsed.usage };
401
+ // v0.6: the CLI's own stop_reason and the modelUsage map's canonical ids.
402
+ return { text: String(parsed.text || '').trim(), usage: parsed.usage, stopReason: parsed.stopReason, reportedModels: parsed.reportedModels, isolation: isolationOf(plan) };
382
403
  }
383
404
 
384
405
  // ── openai/api (Chat Completions-compatible) ───────────────────────────────────
@@ -418,7 +439,8 @@ async function completeOpenaiApi({ system, prompt, model, maxTokens, temperature
418
439
  let parsed;
419
440
  try { parsed = JSON.parse(raw); } catch (_e) { throw new Error(`openai/api: unparseable response: ${raw.slice(0, 200)}`); }
420
441
  const text = ((parsed.choices && parsed.choices[0] && parsed.choices[0].message && parsed.choices[0].message.content) || '');
421
- return { text: String(text).trim(), usage: parseOpenaiApiUsage(parsed.usage) };
442
+ // v0.6: the response's model and the first choice's finish_reason.
443
+ return { text: String(text).trim(), usage: parseOpenaiApiUsage(parsed.usage), ...readOpenaiApiResponse(parsed), isolation: 'none' };
422
444
  }
423
445
 
424
446
  // Read the OpenAI provider config (base_url + api-key env) from the registry,
@@ -515,7 +537,9 @@ async function completeCodexCli({ system, prompt, model, timeoutMs, trusted = fa
515
537
  throw new Error(hop || `codex exec exited ${r.code}: ${String(r.stderr).slice(0, 400)}`);
516
538
  }
517
539
  const ev = parseCodexJsonl(r.stdout);
518
- return { text: String(ev.text || '').trim(), usage: ev.usage };
540
+ // codex reports neither the serving model nor a stop reason on its stream;
541
+ // both read null and the receipt says the surface did not report them.
542
+ return { text: String(ev.text || '').trim(), usage: ev.usage, stopReason: ev.stopReason, reportedModels: ev.reportedModels, isolation: isolationOf(plan) };
519
543
  }
520
544
 
521
545
  module.exports = {
@@ -525,5 +549,5 @@ module.exports = {
525
549
  CODEX_OVERHEAD_NOTE, CODEX_EXEC_ARGS, buildCodexArgs, codexFinalArgs, codexAuthPresent,
526
550
  // spec 022 — the eval-user hop
527
551
  SUDO_BIN, EVAL_USER_DEFAULT, ISOLATED_ENV_ALLOWLIST, ISOLATED_WRAPPER, HOP_LABEL,
528
- evalUser, isolatedEnv, buildSpawnPlan, runIsolated,
552
+ evalUser, isolatedEnv, buildSpawnPlan, runIsolated, isolationOf,
529
553
  };
package/lib/receipt.js CHANGED
@@ -20,9 +20,30 @@ const SCHEMA_FILES = {
20
20
  // published v0.4 receipt asserts conformance by NUMBER, and the number has to
21
21
  // keep resolving to the schema it meant (AC-10).
22
22
  '0.4': 'receipt.v0.4.schema.json',
23
- '0.5': 'receipt.schema.json',
23
+ // v0.5 moved the same way when v0.6 took the current pointer (spec 026): the
24
+ // six report-008 receipts assert v0.5 by number, and v0.6 is not additive for
25
+ // the validator (it requires the receipt to say what answered it), so the
26
+ // number has to keep resolving to the schema it meant.
27
+ '0.5': 'receipt.v0.5.schema.json',
28
+ '0.6': 'receipt.schema.json',
24
29
  };
25
30
 
31
+ // THE BAND RULE, stated on every receipt (results.aggregates.band_rule, spec
32
+ // 026 AC-6) so a reader recomputes the summary from the rows by the formula
33
+ // it names rather than by guessing which of the two "bands" a receipt carries.
34
+ // Per case, stddev is the spread of judge samples; per arm, it is this.
35
+ const BAND_RULE = 'per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases';
36
+
37
+ // THE ONE PREDICATE for "this case did not complete" (spec 026 AC-3). Every
38
+ // reader in lib/ and scripts/ asks this, never the failed_timeout literal: a
39
+ // reader that excluded by that literal admitted a failed_unmeasured case (an
40
+ // empty generation, a judge with no score) into its statistics as if it had
41
+ // been measured. A case is failed when its case_status is present and is not
42
+ // 'ok'; the status names what was observed, and both failure statuses are
43
+ // recorded without fabricated samples or hashes.
44
+ const FAILED_STATUSES = ['failed_timeout', 'failed_unmeasured'];
45
+ function caseFailed(c) { return !!(c && c.case_status && c.case_status !== 'ok'); }
46
+
26
47
  const _validators = {};
27
48
  // Lazily compile the JSON Schema validator (ajv) for a given version. Kept lazy
28
49
  // so the library can be required without ajv present (pure hashing utilities).
@@ -80,6 +101,27 @@ function aggregate(caseResults) {
80
101
  return { case_count: caseResults.length, pass_count: passes, borderline_count: borderline, mean_score: band.mean, stddev: band.stddev };
81
102
  }
82
103
 
104
+ // The comparison block from the two arms' aggregates. Every null it carries is
105
+ // named: `delta_uncertainty_unavailable` is `no_cases` when an arm has no
106
+ // included case (then every score is null too: the mean of nothing is not a
107
+ // number) and `single_case` when an arm has one (a mean, no band). Otherwise
108
+ // the block is numeric and the field is absent (spec 026 AC-7).
109
+ function comparisonOf(aggWith, aggBase) {
110
+ const noCases = aggWith.case_count === 0 || aggBase.case_count === 0;
111
+ const single = !noCases && (aggWith.case_count < 2 || aggBase.case_count < 2);
112
+ const cmp = {
113
+ with_skill_score: aggWith.mean_score,
114
+ baseline_score: aggBase.mean_score,
115
+ delta: noCases ? null : round(aggWith.mean_score - aggBase.mean_score),
116
+ // Combined uncertainty of the delta: quadrature sum of the two aggregate
117
+ // bands. Diff uses this for the headline "within noise" vs real-move rule.
118
+ delta_uncertainty: noCases ? null : combineUncertainty(aggWith.stddev, aggBase.stddev),
119
+ };
120
+ if (noCases) cmp.delta_uncertainty_unavailable = 'no_cases';
121
+ else if (single) cmp.delta_uncertainty_unavailable = 'single_case';
122
+ return cmp;
123
+ }
124
+
83
125
  // Assemble a full receipt from the runner's raw pieces, seal it, and return it.
84
126
  // skill: { name, version, contentHash }
85
127
  // suite: { format, suiteHash, caseCount }
@@ -109,7 +151,7 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
109
151
  // The excluded case STAYS in `results.cases`. It is removed from the mean, not
110
152
  // from the record — deleting the evidence of a failure is a different and worse
111
153
  // defect than averaging over it.
112
- const armUnusable = (c) => c.case_status === 'failed_timeout'
154
+ const armUnusable = (c) => caseFailed(c)
113
155
  || (c.mean == null && c.score == null);
114
156
  const excludedIds = new Map();
115
157
  for (const c of cases) {
@@ -183,25 +225,23 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
183
225
  registry: run.registry || 'unregistered',
184
226
  transcripts: run.transcripts || 'hashes-only',
185
227
  judge: run.judge || { samples: 1, temperature: null, sampling: 'single', surface: run.surface },
228
+ // v0.6 (spec 026 AC-1, AC-2): what answered. Carried from the caller as
229
+ // given and never defaulted: a receipt that does not say what answered it
230
+ // is refused by the schema, which is the point.
231
+ ...(run.answered_by ? { answered_by: run.answered_by } : {}),
186
232
  },
187
233
  results: {
188
234
  cases,
189
235
  aggregates: {
190
236
  with_skill: aggWith,
191
237
  baseline: aggBase,
238
+ band_rule: BAND_RULE,
192
239
  // Present only when something was excluded, so a clean run's receipt is
193
240
  // unchanged and the archive does not acquire an empty field.
194
241
  ...(excludedCases.length ? { excluded_cases: excludedCases } : {}),
195
242
  },
196
243
  },
197
- comparison: {
198
- with_skill_score: aggWith.mean_score,
199
- baseline_score: aggBase.mean_score,
200
- delta: round(aggWith.mean_score - aggBase.mean_score),
201
- // Combined uncertainty of the delta: quadrature sum of the two aggregate
202
- // bands. Diff uses this for the headline "within noise" vs real-move rule.
203
- delta_uncertainty: combineUncertainty(aggWith.stddev, aggBase.stddev),
204
- },
244
+ comparison: comparisonOf(aggWith, aggBase),
205
245
  verification_level: verificationLevel,
206
246
  receipt_hash: '',
207
247
  };
@@ -226,5 +266,5 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
226
266
  }
227
267
 
228
268
  module.exports = {
229
- buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt,
269
+ buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt, BAND_RULE, comparisonOf, caseFailed, FAILED_STATUSES,
230
270
  };