driftproof 0.10.1 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/diff.js CHANGED
@@ -1,16 +1,18 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
- const { bandVerdict, round } = require('./stats');
4
+ const { bandVerdict, round, WITHIN_NOISE } = require('./stats');
5
5
  const { EFFECT_FLOOR } = require('../config');
6
6
  const { revisionHeadline } = require('./revision');
7
7
  const { baselineReproduces, REFUSAL_REASONS, bandOf } = require('./reuse');
8
+ const { caseRule, armOf, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
8
9
 
9
10
  // Practical-significance gate applied ON TOP of band separation. bandVerdict()
10
11
  // stays a pure geometry test (kept that way so its unit checks are unambiguous);
11
12
  // this wrapper additionally requires |delta| >= EFFECT_FLOOR before a verdict is
12
- // claimed. A separated-but-trivial move (below the judge's quantization floor)
13
- // becomes "within noise (below effect floor)".
13
+ // claimed. A separated but trivial move (below the judge's quantization floor)
14
+ // gets the verdict value WITHIN_NOISE_FLOOR, which the table labels "below effect
15
+ // floor".
14
16
  const WITHIN_NOISE_FLOOR = 'within noise (below effect floor)';
15
17
  function verdictWithFloor(before, after, delta) {
16
18
  const raw = bandVerdict(before.mean, before.stddev, after.mean, after.stddev);
@@ -19,17 +21,18 @@ function verdictWithFloor(before, after, delta) {
19
21
  }
20
22
  return raw;
21
23
  }
22
- function isWithinNoise(v) { return v === 'within noise' || v === WITHIN_NOISE_FLOOR; }
24
+ function isWithinNoise(v) { return v === WITHIN_NOISE || v === WITHIN_NOISE_FLOOR; }
23
25
 
24
26
  // Build a drift report (markdown) between two receipts.
25
27
  //
26
- // CREDIBILITY CORE — the anti-false-positive rule: a per-case regression (or
27
- // improvement) is claimed ONLY when (1) the two confidence bands do NOT overlap
28
- // AND (2) the mean moved by at least EFFECT_FLOOR (practical-significance floor,
29
- // see config.js). When the bands overlap OR the move is below the floor, the
30
- // change is reported as "within noise" and is NOT counted as a regression. A tool
31
- // that cries wolf is worse than useless, so band separation PLUS a real-sized
32
- // delta — not either alone — is what triggers a verdict.
28
+ // CREDIBILITY CORE, the anti-false-positive rule: a per-case regression (or
29
+ // improvement) is claimed ONLY when (1) the two bands, each the mean plus or
30
+ // minus one sample standard deviation, do NOT overlap AND (2) the mean moved by at
31
+ // least EFFECT_FLOOR (practical-significance floor, see config.js). When the
32
+ // bands overlap OR the move is below the floor, no separation is detected at the
33
+ // sample size used, and the case is NOT counted as a regression. A tool that cries
34
+ // wolf is worse than useless, so band separation PLUS a real-sized delta, not
35
+ // either alone, is what triggers a verdict.
33
36
 
34
37
  // Map case-id → { mean, stddev } for a receipt's with_skill cases. Falls back to
35
38
  // score/0 for v0.1 receipts that have no per-case band.
@@ -124,14 +127,21 @@ function judgeDiffers(a, b, ids, aB, bB) {
124
127
  // It deliberately does NOT run a separate band test on the aggregate mean: the
125
128
  // aggregate band is suite dispersion, and a separate test there would either cry
126
129
  // wolf (if too tight) or mask real per-case drift (if too wide).
130
+ //
131
+ // THE WORDING (spec 031 A-031-20). A band is a descriptive spread, the mean plus or
132
+ // minus one sample standard deviation. A separation is stated as detected under
133
+ // the rule, never as proof that the skill moved; no separation is stated as none
134
+ // detected at this sample size, never as evidence that nothing changed. The
135
+ // leading word is the label; the verdict values it summarises are unchanged.
127
136
  function headlineVerdict(perCase) {
128
137
  const reg = perCase.filter((r) => r.verdict === 'regression').length;
129
138
  const imp = perCase.filter((r) => r.verdict === 'improvement').length;
130
139
  const s = (n) => (n === 1 ? '' : 's');
131
- if (reg && imp) return `MIXED — ${reg} case regression${s(reg)} and ${imp} improvement${s(imp)} on non-overlapping bands.`;
132
- if (reg) return `DRIFT — ${reg} case${s(reg)} regressed (bands do not overlap); the skill is measurably weaker on ${reg} case${s(reg)}.`;
133
- if (imp) return `IMPROVED — ${imp} case${s(imp)} improved (bands do not overlap); none regressed.`;
134
- return 'WITHIN NOISE — no case moved beyond its confidence band; the skill holds up.';
140
+ const rule = 'bands do not overlap and the move clears the effect floor';
141
+ if (reg && imp) return `MIXED: separation detected under the rule on ${reg} case${s(reg)} downward and ${imp} case${s(imp)} upward (${rule}).`;
142
+ if (reg) return `DRIFT: separation detected under the rule on ${reg} case${s(reg)}, downward (${rule}); none upward.`;
143
+ if (imp) return `IMPROVED: separation detected under the rule on ${imp} case${s(imp)}, upward (${rule}); none downward.`;
144
+ return 'NO SEPARATION DETECTED: no case separated under the rule at this sample size (band = mean \u00b1 1 sd); this is not evidence that nothing changed.';
135
145
  }
136
146
 
137
147
  // A revision pair is a pair in which the SKILL TEXT is the only thing that
@@ -189,6 +199,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
189
199
  // untested input does: no delta is asserted, and the reason travels with it.
190
200
  const measured = belowTested.length === 0 && !refusal && incompleteSides.length === 0 && !judgeProblem;
191
201
 
202
+ // THE UNDERPOWERED STATE RIDES BESIDE THE VERDICT VALUE (spec 035 R-6). The value
203
+ // is unchanged, because Reports 006 to 008 and spec 009's gate compare against it
204
+ // and a published record quotes the lines it produced. `power` says, for a case
205
+ // the rule did not separate, whether the two with_skill arms' own spreads and
206
+ // draws could have resolved a shift of the effect floor, by the rule lib/verdict.js
207
+ // applies to a single receipt.
208
+ const rawWith = (r) => new Map(r.results.cases.filter((c) => c.mode === 'with_skill').map((c) => [c.id, c]));
209
+ const aRaw = rawWith(a);
210
+ const bRaw = rawWith(b);
192
211
  const perCase = ids.map((id) => {
193
212
  const before = aB[id] || null;
194
213
  const after = bB[id] || null;
@@ -196,10 +215,16 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
196
215
  const verdict = refusal ? 'refused'
197
216
  : !measured ? 'not measured'
198
217
  : (before && after) ? verdictWithFloor(before, after, delta) : 'n/a';
199
- return { id, before, after, delta, verdict };
218
+ let power = null;
219
+ if (measured && before && after && (verdict === WITHIN_NOISE || verdict === WITHIN_NOISE_FLOOR)) {
220
+ const ra = armOf(aRaw.get(id)); const rb = armOf(bRaw.get(id));
221
+ const rule = ra && rb ? caseRule(ra, rb) : null;
222
+ if (rule && rule.state === 'underpowered') power = { state: 'underpowered', drawsNeeded: rule.drawsNeeded, reason: rule.reason, ...(rule.spread !== undefined ? { spread: rule.spread } : {}) };
223
+ }
224
+ return { id, before, after, delta, verdict, power };
200
225
  });
201
226
  // Sort worst-first: regressions, then by delta.
202
- const order = { regression: 0, 'within noise': 1, [WITHIN_NOISE_FLOOR]: 1, improvement: 2, 'n/a': 3, 'not measured': 3, refused: 3 };
227
+ const order = { regression: 0, [WITHIN_NOISE]: 1, [WITHIN_NOISE_FLOOR]: 1, improvement: 2, 'n/a': 3, 'not measured': 3, refused: 3 };
203
228
  perCase.sort((x, y) => (order[x.verdict] - order[y.verdict]) || ((x.delta || 0) - (y.delta || 0)));
204
229
 
205
230
  const aAgg = aggWithBand(a);
@@ -310,10 +335,33 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
310
335
  // carries a source. A legend explains markers on the page; if `bandStr` emits
311
336
  // none, the legend is describing something the reader cannot see. Asked of
312
337
  // `bandStr` itself rather than recomputed, so the two cannot disagree.
313
- const anyLabelRendered = [...Object.values(aB), ...Object.values(bB)]
314
- .some((x) => x && /\([a-z]+\)\s*$/.test(bandStr(x)));
315
- if (anyLabelRendered) {
316
- warnings.push('band provenance: `(generation)` is an ACROSS-DRAW spread — receipt spec v0.5, n generation draws per arm. `(legacy)` is a JUDGE-SAMPLE spread over a single generation, which is what v0.4 and earlier recorded. They are different statistics. The comparison is the only one the older receipt admits, and it is not like for like.');
338
+ //
339
+ // THE LEGEND DESCRIBES ONLY THE SOURCES ON THE PAGE, AND THE PAIR IS CALLED
340
+ // NOT LIKE FOR LIKE ONLY WHEN IT IS NOT (spec 031 AC-6). This used to print one
341
+ // sentence whenever any label rendered, naming both sources and saying the
342
+ // comparison was the only one an older receipt admits and was not like for
343
+ // like. Every v0.4 and v0.5 band carries a label, so two v0.5 receipts with
344
+ // `generation` bands on both sides were told they were not like for like, and
345
+ // shown a `(legacy)` source neither carried. The sources are read per receipt,
346
+ // from the markers each side renders; the difference is stated only when the
347
+ // two sides' sets differ, naming which side carries which.
348
+ const renderedSources = (bands) => [...new Set(Object.values(bands)
349
+ .map((x) => (x && (/\(([a-z]+)\)\s*$/.exec(bandStr(x)) || [])[1]) || null).filter(Boolean))].sort();
350
+ const sourcesA = renderedSources(aB);
351
+ const sourcesB = renderedSources(bB);
352
+ const sourcesOnPage = [...new Set([...sourcesA, ...sourcesB])].sort();
353
+ const sourcesDiffer = sourcesA.join() !== sourcesB.join();
354
+ if (sourcesOnPage.length) {
355
+ const LEGEND = {
356
+ generation: '`(generation)` is an ACROSS-DRAW spread: the standard deviation of the case\'s score across n generation draws per arm.',
357
+ legacy: '`(legacy)` is a JUDGE-SAMPLE spread over a single generation: the standard deviation across the judge\'s samples of that one text.',
358
+ };
359
+ const legend = sourcesOnPage.map((s) => LEGEND[s] || `\`(${s})\` is a band source this differ does not describe.`);
360
+ if (sourcesDiffer) {
361
+ const side = (label, s) => `${label} carries ${s.length ? s.map((x) => `\`(${x})\``).join(' and ') : 'no labelled'} bands`;
362
+ legend.push(`The two receipts' bands come from different sources: ${side(labelA, sourcesA)}, ${side(labelB, sourcesB)}. They are different statistics, so a per-case comparison across them is not like for like.`);
363
+ }
364
+ warnings.push(`band provenance: ${legend.join(' ')}`);
317
365
  }
318
366
  if (warnings.length) {
319
367
  L.push('> **⚠ Caveats**');
@@ -350,8 +398,27 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
350
398
  } else {
351
399
  L.push(`**${revision ? revisionHeadline(perCase) : headlineVerdict(perCase)}**`);
352
400
  L.push('');
353
- L.push(`with_skill mean moved ${fmt(headlineDelta)} (${bandStr(aAgg)} → ${bandStr(bAgg)}; band = suite dispersion). Per-case band-overlap verdicts: ${regressions.length} regression(s), ${perCase.filter((r) => r.verdict === 'improvement').length} improvement(s), ${nWithin} within noise${nFloor ? ` (${nFloor} of them band-separated but below the ${EFFECT_FLOOR} effect floor)` : ''}.`);
401
+ // ONE LABEL PER CASE, THE TABLE'S (F-5 of
402
+ // specs/031-artefact-claims/evidence/approval-20260915T032341Z.md). A case whose
403
+ // bands do not overlap but whose move is below the floor is labelled below
404
+ // effect floor in the per-case table. This line counted it among the cases with
405
+ // no separation detected and then called the same cases band-separated; it now
406
+ // counts the two apart, as the table labels them. Neither is a separation under
407
+ // the rule, which is what the headline above says.
408
+ L.push(`with_skill mean moved ${fmt(headlineDelta)} (${bandStr(aAgg)} → ${bandStr(bAgg)}; band = suite dispersion). Per-case band-overlap verdicts: ${regressions.length} regression(s), ${perCase.filter((r) => r.verdict === 'improvement').length} improvement(s), ${nWithin - nFloor} with no separation detected${nFloor ? `, ${nFloor} below the ${EFFECT_FLOOR} effect floor` : ''}.`);
354
409
  L.push('');
410
+ const underpowered = perCase.filter((r) => r.power);
411
+ if (underpowered.length) {
412
+ // One count, one sentence, one draws line: the worst case's (spec 035 AC-5).
413
+ const spread = underpowered.find((r) => r.power.reason === 'spread');
414
+ const most = underpowered.filter((r) => r.power.reason === 'draws').sort((x, y) => y.power.drawsNeeded - x.power.drawsNeeded)[0];
415
+ const worst = spread ? { value: null, reason: 'spread', case: spread.id, spread: spread.power.spread }
416
+ : most ? { value: most.power.drawsNeeded, reason: 'draws', case: most.id }
417
+ : { value: 2, reason: 'single_draw', case: underpowered[0].id };
418
+ const powerLine = `Underpowered: ${underpowered.length} of the cases with no separation under the rule. ${UNDERPOWERED_LINE}. ${drawsLine(worst)}`;
419
+ L.push(powerLine);
420
+ L.push('');
421
+ }
355
422
  }
356
423
 
357
424
  L.push(`## Per-case with_skill (band overlap → verdict)`);
@@ -359,8 +426,9 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
359
426
  L.push(`| case | ${labelA} (mean ± sd) | ${labelB} (mean ± sd) | Δ | verdict |`);
360
427
  L.push(`|---|---|---|---|---|`);
361
428
  for (const r of perCase) {
362
- const flag = r.verdict === 'regression' ? '🔻 regression' : r.verdict === 'improvement' ? '🔼 improvement' : r.verdict === WITHIN_NOISE_FLOOR ? 'within noise (below floor)' : r.verdict === 'within noise' ? 'within noise' : r.verdict === 'not measured' ? 'not measured' : 'n/a';
363
- L.push(`| \`${r.id}\` | ${r.before ? bandStr(r.before) : 'n/a'} | ${r.after ? bandStr(r.after) : 'n/a'} | ${fmt(r.delta)} | ${flag} |`);
429
+ const flag = r.verdict === 'regression' ? '🔻 regression' : r.verdict === 'improvement' ? '🔼 improvement' : r.verdict === WITHIN_NOISE_FLOOR ? 'below effect floor' : r.verdict === WITHIN_NOISE ? 'no separation detected' : r.verdict === 'not measured' ? 'not measured' : 'n/a';
430
+ const label = r.power ? `${flag}; ${UNDERPOWERED_LINE}` : flag;
431
+ L.push(`| \`${r.id}\` | ${r.before ? bandStr(r.before) : 'n/a'} | ${r.after ? bandStr(r.after) : 'n/a'} | ${fmt(r.delta)} | ${label} |`);
364
432
  }
365
433
  L.push('');
366
434
 
package/lib/judge.js CHANGED
@@ -78,8 +78,15 @@ function promptTemplateHash() {
78
78
  const TRUNCATION_STOP_REASONS = new Set(['max_tokens', 'length']);
79
79
  function isTruncated(stopReason) { return TRUNCATION_STOP_REASONS.has(String(stopReason || '')); }
80
80
 
81
+ // Only a JSON number is a score (spec 032 AC-1). Number() turned null, false,
82
+ // "" and [] into 0 and true into 1, and each of those became a measured sample:
83
+ // a judge that said nothing scored the worst possible grade, and two receipts
84
+ // over the same generations read as a regression. Every non-number becomes NaN
85
+ // here and is refused by the isFinite line below, which stays the one line every
86
+ // non-number reaches (spec 026's clamp mutation is planted on it).
81
87
  function scoreOf(parsed) {
82
- const x = Number(parsed && parsed.score);
88
+ const raw = parsed ? parsed.score : undefined;
89
+ const x = typeof raw === 'number' ? raw : NaN;
83
90
  if (!Number.isFinite(x)) return null;
84
91
  if (x < 0 || x > 1) return null;
85
92
  return x;
@@ -128,10 +135,11 @@ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature
128
135
  }
129
136
  const score = scoreOf(parsed);
130
137
  if (score === null) {
131
- const x = parsed && parsed.score;
132
- const reason = x === undefined || x === null ? 'judge output carries no numeric score'
133
- : !Number.isFinite(Number(x)) ? `judge output carries no numeric score (score ${JSON.stringify(x)})`
134
- : `judge score ${x} outside [0, 1]`;
138
+ const x = parsed ? parsed.score : undefined;
139
+ const reason = x === undefined ? 'judge output carries no numeric score'
140
+ : typeof x !== 'number' ? `judge output carries no numeric score (score ${JSON.stringify(x)})`
141
+ : !Number.isFinite(x) ? `judge output carries no numeric score (score ${String(x)})`
142
+ : `judge score ${x} outside [0, 1]`;
135
143
  return { ...base, unmeasured: true, reason };
136
144
  }
137
145
  return { ...base, score, reason: String(parsed.reason || '').slice(0, 300) };
@@ -143,7 +151,7 @@ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature
143
151
  // samples: a partial sample set must never become a band, and a draw one of
144
152
  // whose judge samples carried no score is unmeasured as a whole (spec 026
145
153
  // AC-3). The remaining samples are not taken.
146
- // `mean` ± `stddev` is the per-case confidence band (raw spread of the N scores)
154
+ // `mean` ± `stddev` is the per-case band, a descriptive spread (one sample sd of the N scores)
147
155
  // used by the borderline-outcome rule and per-case drift band-overlap logic.
148
156
  // NO DEFAULT TIMEOUT HERE (spec 017 AC-2). This defaulted to 120000, which
149
157
  // outranked the per-surface policy exactly as lib/run.js's literal did — so the
package/lib/receipt.js CHANGED
@@ -25,7 +25,11 @@ const SCHEMA_FILES = {
25
25
  // the validator (it requires the receipt to say what answered it), so the
26
26
  // number has to keep resolving to the schema it meant.
27
27
  '0.5': 'receipt.v0.5.schema.json',
28
- '0.6': 'receipt.schema.json',
28
+ // v0.6 moved the same way when v0.7 took the current pointer (spec 035): v0.7 adds a
29
+ // verdict token a reader derives (UNDERPOWERED), and a v0.6 receipt restamped 0.7
30
+ // is refused by the const, so the number keeps resolving to the schema it meant.
31
+ '0.6': 'receipt.v0.6.schema.json',
32
+ '0.7': 'receipt.schema.json',
29
33
  };
30
34
 
31
35
  // THE BAND RULE, stated on every receipt (results.aggregates.band_rule, spec
@@ -114,7 +118,7 @@ function comparisonOf(aggWith, aggBase) {
114
118
  baseline_score: aggBase.mean_score,
115
119
  delta: noCases ? null : round(aggWith.mean_score - aggBase.mean_score),
116
120
  // Combined uncertainty of the delta: quadrature sum of the two aggregate
117
- // bands. Diff uses this for the headline "within noise" vs real-move rule.
121
+ // bands. A figure beside the delta; the per-case band rule decides the headline.
118
122
  delta_uncertainty: noCases ? null : combineUncertainty(aggWith.stddev, aggBase.stddev),
119
123
  };
120
124
  if (noCases) cmp.delta_uncertainty_unavailable = 'no_cases';
package/lib/revision.js CHANGED
@@ -21,29 +21,38 @@ const { EFFECT_FLOOR } = require('../config');
21
21
 
22
22
  // ── the cell headline ────────────────────────────────────────────────────────
23
23
  // A summary of the per-case band-overlap verdicts, worded about the REVISION.
24
- // The release-drift headline says "the skill is measurably weaker", which is a
25
- // sentence about a skill under a moving model. Here the model is the control.
24
+ // The release-drift headline is a sentence about a skill under a moving model.
25
+ // Here the model is the control. Worded under spec 031 A-031-20: a separation is
26
+ // detected under the rule, never proof; none detected is not evidence of sameness.
26
27
  function revisionHeadline(perCase) {
27
28
  const reg = perCase.filter((r) => r.verdict === 'regression').length;
28
29
  const imp = perCase.filter((r) => r.verdict === 'improvement').length;
29
30
  const s = (n) => (n === 1 ? '' : 's');
30
31
  if (reg && imp) {
31
- return `MIXED — the revision improved ${imp} case${s(imp)} and regressed ${reg} on non-overlapping bands.`;
32
+ return `MIXED: separation detected under the rule on ${imp} case${s(imp)} upward and ${reg} downward under the current upstream text (bands do not overlap).`;
32
33
  }
33
34
  if (reg) {
34
- return `REVISION REGRESSED — ${reg} case${s(reg)} scored lower under the current upstream text (bands do not overlap).`;
35
+ return `REVISION REGRESSED: separation detected under the rule on ${reg} case${s(reg)}, lower under the current upstream text (bands do not overlap).`;
35
36
  }
36
37
  if (imp) {
37
- return `REVISION IMPROVED — ${imp} case${s(imp)} scored higher under the current upstream text (bands do not overlap); none regressed.`;
38
+ return `REVISION IMPROVED: separation detected under the rule on ${imp} case${s(imp)}, higher under the current upstream text (bands do not overlap); none lower.`;
38
39
  }
39
- return 'WITHIN NOISE — the revision moved no case beyond its confidence band; the pinned text and the current text measure the same.';
40
+ return 'NO SEPARATION DETECTED: no case separated under the rule at this sample size (band = mean \u00b1 1 sd); this is not evidence that the pinned text and the current text behave alike.';
40
41
  }
41
42
 
42
- // Classification word for a cell, from its headline. Kept separate so a caller
43
- // can branch on the class without parsing prose.
43
+ // Classification value for a cell. Kept separate so a caller can branch on the
44
+ // class without parsing prose. It used to be the headline's first word, cut at
45
+ // its dash; the headline's wording is now a display matter (spec 031 A-031-20),
46
+ // so the values are stated here and are unchanged: scripts/prepare-report-006.js
47
+ // and spec 009's gate read them. CLASS_NOT_SEPARATED is the value, not a label.
48
+ const CLASS_NOT_SEPARATED = 'WITHIN NOISE';
44
49
  function revisionClass(perCase) {
45
- const h = revisionHeadline(perCase);
46
- return h.split(' —')[0];
50
+ const reg = perCase.filter((r) => r.verdict === 'regression').length;
51
+ const imp = perCase.filter((r) => r.verdict === 'improvement').length;
52
+ if (reg && imp) return 'MIXED';
53
+ if (reg) return 'REVISION REGRESSED';
54
+ if (imp) return 'REVISION IMPROVED';
55
+ return CLASS_NOT_SEPARATED;
47
56
  }
48
57
 
49
58
  // ── the fairness sentence ────────────────────────────────────────────────────
@@ -54,7 +63,7 @@ function revisionClass(perCase) {
54
63
  // one-directional courtesy — the symmetry is a property of the code, and the
55
64
  // gate asserts it.
56
65
  //
57
- // A cell within noise gets NO sentence. #005's figure stands unamended, because
66
+ // A cell with no separation detected gets NO sentence. #005's figure stands unamended, because
58
67
  // nothing was measured that would amend it, and manufacturing a hedge for a null
59
68
  // result is how a report launders noise into a finding.
60
69
  function fairnessSentence({ slug, classification, report005Delta, measuredDelta }) {
package/lib/stats.js CHANGED
@@ -1,8 +1,9 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
- // Small statistics helpers for sampled judging and confidence bands. Kept
5
- // dependency-free and deterministic.
4
+ // Small statistics helpers for sampled scores and their bands. A band is a
5
+ // descriptive spread, the mean plus or minus one sample standard deviation; it
6
+ // carries no coverage probability. Kept dependency-free and deterministic.
6
7
 
7
8
  function round(n, dp = 6) { const f = Math.pow(10, dp); return Math.round(n * f) / f; }
8
9
 
@@ -60,16 +61,24 @@ function aggregateBands(cases) {
60
61
  return { mean: mean(means), stddev: means.length < 2 ? null : stddev(means) };
61
62
  }
62
63
 
63
- // Do two confidence bands (mean ± half-width) fail to overlap, and in which
64
- // direction? Returns 'regression' (b below a), 'improvement' (b above a), or
65
- // 'within noise' (bands touch/overlap). This is the anti-false-positive rule:
66
- // a change is only claimed when the bands are fully separated.
64
+ // THE NOT-SEPARATED VERDICT VALUE, which is a value and not a label (spec 031
65
+ // A-031-20). Callers compare against it, and every display label is mapped from
66
+ // it where it is rendered: lib/diff.js prints "no separation detected". The value
67
+ // itself is unchanged pending the verdict-rule spec.
68
+ const WITHIN_NOISE = 'within noise';
69
+
70
+ // Do two bands (mean ± half-width, the half-width one sample standard deviation)
71
+ // fail to overlap, and in which direction? Returns 'regression' (b below a),
72
+ // 'improvement' (b above a), or WITHIN_NOISE (the bands touch or overlap). This
73
+ // is the anti-false-positive rule: a separation is detected only when the bands
74
+ // are fully apart, and overlap is the absence of a detected separation at the
75
+ // sample size used, not evidence that nothing changed.
67
76
  // regression : meanB + hwB < meanA - hwA
68
77
  // improvement : meanB - hwB > meanA + hwA
69
78
  function bandVerdict(meanA, hwA, meanB, hwB) {
70
79
  if (meanB + hwB < meanA - hwA) return 'regression';
71
80
  if (meanB - hwB > meanA + hwA) return 'improvement';
72
- return 'within noise';
81
+ return WITHIN_NOISE;
73
82
  }
74
83
 
75
84
  // The variance ratio: how much larger the GENERATION-level spread is than the
@@ -87,4 +96,4 @@ function varianceRatio(generationSd, judgeSdMean) {
87
96
  return round(g / j);
88
97
  }
89
98
 
90
- module.exports = { mean, stddev, stderr, combineUncertainty, aggregateBands, bandVerdict, round, varianceRatio };
99
+ module.exports = { mean, stddev, stderr, combineUncertainty, aggregateBands, bandVerdict, round, varianceRatio, WITHIN_NOISE };
package/lib/value.js CHANGED
@@ -25,8 +25,8 @@ const { EFFECT_FLOOR } = require('../config');
25
25
  //
26
26
  // RATIO FRAMINGS ARE FLOOR-GATED. A ratio like "dollars per 0.01 lift" is
27
27
  // only meaningful when the benefit it prices is a real move. When the lift did not clear
28
- // the effect floor (or its bands overlap), the ratio renders "n/a (within noise)"
29
- // — never a number, however tempting the arithmetic. Dividing noise by a cost
28
+ // the effect floor (or its bands overlap), the ratio renders "n/a (no separation
29
+ // detected)", never a number, however tempting the arithmetic. Dividing noise by a cost
30
30
  // produces a precise-looking figure with no evidence under it.
31
31
  //
32
32
  // PRICING IS FROZEN AT RUN TIME. Registry prices change (vendors cut prices; we
@@ -47,14 +47,14 @@ const COST_BASIS = {
47
47
  const LATENCY_DISCLOSURE =
48
48
  'observed on subscription CLI surface, indicative';
49
49
 
50
- const NOISE_CELL = 'n/a (within noise)';
50
+ const NOISE_CELL = 'n/a (no separation detected)';
51
51
 
52
52
  // A floor-clearing NEGATIVE lift: the skill measurably hurt, so there is no
53
53
  // benefit to put a price on.
54
54
  const REGRESSED_CELL = 'n/a (skill regressed)';
55
55
 
56
56
  // A cell with at least one separated, floor-clearing DRIVER whose AGGREGATE lift
57
- // does not clear the floor. It is not "within noise" — the QA re-derivation found
57
+ // does not clear the floor. It is not the no-separation cell: the QA re-derivation found
58
58
  // six such cells rendering the noise string while the Verdict basis on the same
59
59
  // page named their drivers (QA V-4, 2026-08-19). The aggregate stays the
60
60
  // denominator (spec 002 AC-4, DECISIONS #9), so no price is quoted, but the cell
@@ -461,7 +461,7 @@ function costPerLiftPoint({ lift, separated, incrementalCostPer1kCalls }) {
461
461
  // Two different absences, two different strings. `separated` means at least
462
462
  // one case cleared the floor with non-overlapping bands; when that holds and
463
463
  // only the aggregate falls short, the evidence exists and is listed under
464
- // Verdict basis — saying "within noise" there contradicts the same page.
464
+ // Verdict basis, and the no-separation cell there would contradict the same page.
465
465
  return separated ? DRIVER_ONLY_CELL : NOISE_CELL;
466
466
  }
467
467
  const cost = Number(incrementalCostPer1kCalls);