driftproof 0.10.1 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +47 -28
- package/bin/driftproof +22 -4
- package/config.js +13 -6
- package/lib/badge-svg.js +130 -0
- package/lib/decision.js +25 -8
- package/lib/diff.js +92 -24
- package/lib/judge.js +14 -6
- package/lib/receipt.js +6 -2
- package/lib/revision.js +20 -11
- package/lib/stats.js +17 -8
- package/lib/value.js +5 -5
- package/lib/verdict.js +197 -26
- package/package.json +1 -1
- package/spec/RECEIPT.md +60 -15
- package/spec/receipt.schema.json +2 -2
- package/spec/receipt.v0.6.schema.json +1563 -0
package/lib/diff.js
CHANGED
|
@@ -1,16 +1,18 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
-
const { bandVerdict, round } = require('./stats');
|
|
4
|
+
const { bandVerdict, round, WITHIN_NOISE } = require('./stats');
|
|
5
5
|
const { EFFECT_FLOOR } = require('../config');
|
|
6
6
|
const { revisionHeadline } = require('./revision');
|
|
7
7
|
const { baselineReproduces, REFUSAL_REASONS, bandOf } = require('./reuse');
|
|
8
|
+
const { caseRule, armOf, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
|
|
8
9
|
|
|
9
10
|
// Practical-significance gate applied ON TOP of band separation. bandVerdict()
|
|
10
11
|
// stays a pure geometry test (kept that way so its unit checks are unambiguous);
|
|
11
12
|
// this wrapper additionally requires |delta| >= EFFECT_FLOOR before a verdict is
|
|
12
|
-
// claimed. A separated
|
|
13
|
-
//
|
|
13
|
+
// claimed. A separated but trivial move (below the judge's quantization floor)
|
|
14
|
+
// gets the verdict value WITHIN_NOISE_FLOOR, which the table labels "below effect
|
|
15
|
+
// floor".
|
|
14
16
|
const WITHIN_NOISE_FLOOR = 'within noise (below effect floor)';
|
|
15
17
|
function verdictWithFloor(before, after, delta) {
|
|
16
18
|
const raw = bandVerdict(before.mean, before.stddev, after.mean, after.stddev);
|
|
@@ -19,17 +21,18 @@ function verdictWithFloor(before, after, delta) {
|
|
|
19
21
|
}
|
|
20
22
|
return raw;
|
|
21
23
|
}
|
|
22
|
-
function isWithinNoise(v) { return v ===
|
|
24
|
+
function isWithinNoise(v) { return v === WITHIN_NOISE || v === WITHIN_NOISE_FLOOR; }
|
|
23
25
|
|
|
24
26
|
// Build a drift report (markdown) between two receipts.
|
|
25
27
|
//
|
|
26
|
-
// CREDIBILITY CORE
|
|
27
|
-
// improvement) is claimed ONLY when (1) the two
|
|
28
|
-
// AND (2) the mean moved by at
|
|
29
|
-
// see config.js). When the
|
|
30
|
-
//
|
|
31
|
-
//
|
|
32
|
-
//
|
|
28
|
+
// CREDIBILITY CORE, the anti-false-positive rule: a per-case regression (or
|
|
29
|
+
// improvement) is claimed ONLY when (1) the two bands, each the mean plus or
|
|
30
|
+
// minus one sample standard deviation, do NOT overlap AND (2) the mean moved by at
|
|
31
|
+
// least EFFECT_FLOOR (practical-significance floor, see config.js). When the
|
|
32
|
+
// bands overlap OR the move is below the floor, no separation is detected at the
|
|
33
|
+
// sample size used, and the case is NOT counted as a regression. A tool that cries
|
|
34
|
+
// wolf is worse than useless, so band separation PLUS a real-sized delta, not
|
|
35
|
+
// either alone, is what triggers a verdict.
|
|
33
36
|
|
|
34
37
|
// Map case-id → { mean, stddev } for a receipt's with_skill cases. Falls back to
|
|
35
38
|
// score/0 for v0.1 receipts that have no per-case band.
|
|
@@ -124,14 +127,21 @@ function judgeDiffers(a, b, ids, aB, bB) {
|
|
|
124
127
|
// It deliberately does NOT run a separate band test on the aggregate mean: the
|
|
125
128
|
// aggregate band is suite dispersion, and a separate test there would either cry
|
|
126
129
|
// wolf (if too tight) or mask real per-case drift (if too wide).
|
|
130
|
+
//
|
|
131
|
+
// THE WORDING (spec 031 A-031-20). A band is a descriptive spread, the mean plus or
|
|
132
|
+
// minus one sample standard deviation. A separation is stated as detected under
|
|
133
|
+
// the rule, never as proof that the skill moved; no separation is stated as none
|
|
134
|
+
// detected at this sample size, never as evidence that nothing changed. The
|
|
135
|
+
// leading word is the label; the verdict values it summarises are unchanged.
|
|
127
136
|
function headlineVerdict(perCase) {
|
|
128
137
|
const reg = perCase.filter((r) => r.verdict === 'regression').length;
|
|
129
138
|
const imp = perCase.filter((r) => r.verdict === 'improvement').length;
|
|
130
139
|
const s = (n) => (n === 1 ? '' : 's');
|
|
131
|
-
|
|
132
|
-
if (reg) return `
|
|
133
|
-
if (
|
|
134
|
-
return
|
|
140
|
+
const rule = 'bands do not overlap and the move clears the effect floor';
|
|
141
|
+
if (reg && imp) return `MIXED: separation detected under the rule on ${reg} case${s(reg)} downward and ${imp} case${s(imp)} upward (${rule}).`;
|
|
142
|
+
if (reg) return `DRIFT: separation detected under the rule on ${reg} case${s(reg)}, downward (${rule}); none upward.`;
|
|
143
|
+
if (imp) return `IMPROVED: separation detected under the rule on ${imp} case${s(imp)}, upward (${rule}); none downward.`;
|
|
144
|
+
return 'NO SEPARATION DETECTED: no case separated under the rule at this sample size (band = mean \u00b1 1 sd); this is not evidence that nothing changed.';
|
|
135
145
|
}
|
|
136
146
|
|
|
137
147
|
// A revision pair is a pair in which the SKILL TEXT is the only thing that
|
|
@@ -189,6 +199,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
189
199
|
// untested input does: no delta is asserted, and the reason travels with it.
|
|
190
200
|
const measured = belowTested.length === 0 && !refusal && incompleteSides.length === 0 && !judgeProblem;
|
|
191
201
|
|
|
202
|
+
// THE UNDERPOWERED STATE RIDES BESIDE THE VERDICT VALUE (spec 035 R-6). The value
|
|
203
|
+
// is unchanged, because Reports 006 to 008 and spec 009's gate compare against it
|
|
204
|
+
// and a published record quotes the lines it produced. `power` says, for a case
|
|
205
|
+
// the rule did not separate, whether the two with_skill arms' own spreads and
|
|
206
|
+
// draws could have resolved a shift of the effect floor, by the rule lib/verdict.js
|
|
207
|
+
// applies to a single receipt.
|
|
208
|
+
const rawWith = (r) => new Map(r.results.cases.filter((c) => c.mode === 'with_skill').map((c) => [c.id, c]));
|
|
209
|
+
const aRaw = rawWith(a);
|
|
210
|
+
const bRaw = rawWith(b);
|
|
192
211
|
const perCase = ids.map((id) => {
|
|
193
212
|
const before = aB[id] || null;
|
|
194
213
|
const after = bB[id] || null;
|
|
@@ -196,10 +215,16 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
196
215
|
const verdict = refusal ? 'refused'
|
|
197
216
|
: !measured ? 'not measured'
|
|
198
217
|
: (before && after) ? verdictWithFloor(before, after, delta) : 'n/a';
|
|
199
|
-
|
|
218
|
+
let power = null;
|
|
219
|
+
if (measured && before && after && (verdict === WITHIN_NOISE || verdict === WITHIN_NOISE_FLOOR)) {
|
|
220
|
+
const ra = armOf(aRaw.get(id)); const rb = armOf(bRaw.get(id));
|
|
221
|
+
const rule = ra && rb ? caseRule(ra, rb) : null;
|
|
222
|
+
if (rule && rule.state === 'underpowered') power = { state: 'underpowered', drawsNeeded: rule.drawsNeeded, reason: rule.reason, ...(rule.spread !== undefined ? { spread: rule.spread } : {}) };
|
|
223
|
+
}
|
|
224
|
+
return { id, before, after, delta, verdict, power };
|
|
200
225
|
});
|
|
201
226
|
// Sort worst-first: regressions, then by delta.
|
|
202
|
-
const order = { regression: 0,
|
|
227
|
+
const order = { regression: 0, [WITHIN_NOISE]: 1, [WITHIN_NOISE_FLOOR]: 1, improvement: 2, 'n/a': 3, 'not measured': 3, refused: 3 };
|
|
203
228
|
perCase.sort((x, y) => (order[x.verdict] - order[y.verdict]) || ((x.delta || 0) - (y.delta || 0)));
|
|
204
229
|
|
|
205
230
|
const aAgg = aggWithBand(a);
|
|
@@ -310,10 +335,33 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
310
335
|
// carries a source. A legend explains markers on the page; if `bandStr` emits
|
|
311
336
|
// none, the legend is describing something the reader cannot see. Asked of
|
|
312
337
|
// `bandStr` itself rather than recomputed, so the two cannot disagree.
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
338
|
+
//
|
|
339
|
+
// THE LEGEND DESCRIBES ONLY THE SOURCES ON THE PAGE, AND THE PAIR IS CALLED
|
|
340
|
+
// NOT LIKE FOR LIKE ONLY WHEN IT IS NOT (spec 031 AC-6). This used to print one
|
|
341
|
+
// sentence whenever any label rendered, naming both sources and saying the
|
|
342
|
+
// comparison was the only one an older receipt admits and was not like for
|
|
343
|
+
// like. Every v0.4 and v0.5 band carries a label, so two v0.5 receipts with
|
|
344
|
+
// `generation` bands on both sides were told they were not like for like, and
|
|
345
|
+
// shown a `(legacy)` source neither carried. The sources are read per receipt,
|
|
346
|
+
// from the markers each side renders; the difference is stated only when the
|
|
347
|
+
// two sides' sets differ, naming which side carries which.
|
|
348
|
+
const renderedSources = (bands) => [...new Set(Object.values(bands)
|
|
349
|
+
.map((x) => (x && (/\(([a-z]+)\)\s*$/.exec(bandStr(x)) || [])[1]) || null).filter(Boolean))].sort();
|
|
350
|
+
const sourcesA = renderedSources(aB);
|
|
351
|
+
const sourcesB = renderedSources(bB);
|
|
352
|
+
const sourcesOnPage = [...new Set([...sourcesA, ...sourcesB])].sort();
|
|
353
|
+
const sourcesDiffer = sourcesA.join() !== sourcesB.join();
|
|
354
|
+
if (sourcesOnPage.length) {
|
|
355
|
+
const LEGEND = {
|
|
356
|
+
generation: '`(generation)` is an ACROSS-DRAW spread: the standard deviation of the case\'s score across n generation draws per arm.',
|
|
357
|
+
legacy: '`(legacy)` is a JUDGE-SAMPLE spread over a single generation: the standard deviation across the judge\'s samples of that one text.',
|
|
358
|
+
};
|
|
359
|
+
const legend = sourcesOnPage.map((s) => LEGEND[s] || `\`(${s})\` is a band source this differ does not describe.`);
|
|
360
|
+
if (sourcesDiffer) {
|
|
361
|
+
const side = (label, s) => `${label} carries ${s.length ? s.map((x) => `\`(${x})\``).join(' and ') : 'no labelled'} bands`;
|
|
362
|
+
legend.push(`The two receipts' bands come from different sources: ${side(labelA, sourcesA)}, ${side(labelB, sourcesB)}. They are different statistics, so a per-case comparison across them is not like for like.`);
|
|
363
|
+
}
|
|
364
|
+
warnings.push(`band provenance: ${legend.join(' ')}`);
|
|
317
365
|
}
|
|
318
366
|
if (warnings.length) {
|
|
319
367
|
L.push('> **⚠ Caveats**');
|
|
@@ -350,8 +398,27 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
350
398
|
} else {
|
|
351
399
|
L.push(`**${revision ? revisionHeadline(perCase) : headlineVerdict(perCase)}**`);
|
|
352
400
|
L.push('');
|
|
353
|
-
|
|
401
|
+
// ONE LABEL PER CASE, THE TABLE'S (F-5 of
|
|
402
|
+
// specs/031-artefact-claims/evidence/approval-20260915T032341Z.md). A case whose
|
|
403
|
+
// bands do not overlap but whose move is below the floor is labelled below
|
|
404
|
+
// effect floor in the per-case table. This line counted it among the cases with
|
|
405
|
+
// no separation detected and then called the same cases band-separated; it now
|
|
406
|
+
// counts the two apart, as the table labels them. Neither is a separation under
|
|
407
|
+
// the rule, which is what the headline above says.
|
|
408
|
+
L.push(`with_skill mean moved ${fmt(headlineDelta)} (${bandStr(aAgg)} → ${bandStr(bAgg)}; band = suite dispersion). Per-case band-overlap verdicts: ${regressions.length} regression(s), ${perCase.filter((r) => r.verdict === 'improvement').length} improvement(s), ${nWithin - nFloor} with no separation detected${nFloor ? `, ${nFloor} below the ${EFFECT_FLOOR} effect floor` : ''}.`);
|
|
354
409
|
L.push('');
|
|
410
|
+
const underpowered = perCase.filter((r) => r.power);
|
|
411
|
+
if (underpowered.length) {
|
|
412
|
+
// One count, one sentence, one draws line: the worst case's (spec 035 AC-5).
|
|
413
|
+
const spread = underpowered.find((r) => r.power.reason === 'spread');
|
|
414
|
+
const most = underpowered.filter((r) => r.power.reason === 'draws').sort((x, y) => y.power.drawsNeeded - x.power.drawsNeeded)[0];
|
|
415
|
+
const worst = spread ? { value: null, reason: 'spread', case: spread.id, spread: spread.power.spread }
|
|
416
|
+
: most ? { value: most.power.drawsNeeded, reason: 'draws', case: most.id }
|
|
417
|
+
: { value: 2, reason: 'single_draw', case: underpowered[0].id };
|
|
418
|
+
const powerLine = `Underpowered: ${underpowered.length} of the cases with no separation under the rule. ${UNDERPOWERED_LINE}. ${drawsLine(worst)}`;
|
|
419
|
+
L.push(powerLine);
|
|
420
|
+
L.push('');
|
|
421
|
+
}
|
|
355
422
|
}
|
|
356
423
|
|
|
357
424
|
L.push(`## Per-case with_skill (band overlap → verdict)`);
|
|
@@ -359,8 +426,9 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
359
426
|
L.push(`| case | ${labelA} (mean ± sd) | ${labelB} (mean ± sd) | Δ | verdict |`);
|
|
360
427
|
L.push(`|---|---|---|---|---|`);
|
|
361
428
|
for (const r of perCase) {
|
|
362
|
-
const flag = r.verdict === 'regression' ? '🔻 regression' : r.verdict === 'improvement' ? '🔼 improvement' : r.verdict === WITHIN_NOISE_FLOOR ? '
|
|
363
|
-
|
|
429
|
+
const flag = r.verdict === 'regression' ? '🔻 regression' : r.verdict === 'improvement' ? '🔼 improvement' : r.verdict === WITHIN_NOISE_FLOOR ? 'below effect floor' : r.verdict === WITHIN_NOISE ? 'no separation detected' : r.verdict === 'not measured' ? 'not measured' : 'n/a';
|
|
430
|
+
const label = r.power ? `${flag}; ${UNDERPOWERED_LINE}` : flag;
|
|
431
|
+
L.push(`| \`${r.id}\` | ${r.before ? bandStr(r.before) : 'n/a'} | ${r.after ? bandStr(r.after) : 'n/a'} | ${fmt(r.delta)} | ${label} |`);
|
|
364
432
|
}
|
|
365
433
|
L.push('');
|
|
366
434
|
|
package/lib/judge.js
CHANGED
|
@@ -78,8 +78,15 @@ function promptTemplateHash() {
|
|
|
78
78
|
const TRUNCATION_STOP_REASONS = new Set(['max_tokens', 'length']);
|
|
79
79
|
function isTruncated(stopReason) { return TRUNCATION_STOP_REASONS.has(String(stopReason || '')); }
|
|
80
80
|
|
|
81
|
+
// Only a JSON number is a score (spec 032 AC-1). Number() turned null, false,
|
|
82
|
+
// "" and [] into 0 and true into 1, and each of those became a measured sample:
|
|
83
|
+
// a judge that said nothing scored the worst possible grade, and two receipts
|
|
84
|
+
// over the same generations read as a regression. Every non-number becomes NaN
|
|
85
|
+
// here and is refused by the isFinite line below, which stays the one line every
|
|
86
|
+
// non-number reaches (spec 026's clamp mutation is planted on it).
|
|
81
87
|
function scoreOf(parsed) {
|
|
82
|
-
const
|
|
88
|
+
const raw = parsed ? parsed.score : undefined;
|
|
89
|
+
const x = typeof raw === 'number' ? raw : NaN;
|
|
83
90
|
if (!Number.isFinite(x)) return null;
|
|
84
91
|
if (x < 0 || x > 1) return null;
|
|
85
92
|
return x;
|
|
@@ -128,10 +135,11 @@ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature
|
|
|
128
135
|
}
|
|
129
136
|
const score = scoreOf(parsed);
|
|
130
137
|
if (score === null) {
|
|
131
|
-
const x = parsed
|
|
132
|
-
const reason = x === undefined
|
|
133
|
-
:
|
|
134
|
-
: `judge score ${x}
|
|
138
|
+
const x = parsed ? parsed.score : undefined;
|
|
139
|
+
const reason = x === undefined ? 'judge output carries no numeric score'
|
|
140
|
+
: typeof x !== 'number' ? `judge output carries no numeric score (score ${JSON.stringify(x)})`
|
|
141
|
+
: !Number.isFinite(x) ? `judge output carries no numeric score (score ${String(x)})`
|
|
142
|
+
: `judge score ${x} outside [0, 1]`;
|
|
135
143
|
return { ...base, unmeasured: true, reason };
|
|
136
144
|
}
|
|
137
145
|
return { ...base, score, reason: String(parsed.reason || '').slice(0, 300) };
|
|
@@ -143,7 +151,7 @@ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature
|
|
|
143
151
|
// samples: a partial sample set must never become a band, and a draw one of
|
|
144
152
|
// whose judge samples carried no score is unmeasured as a whole (spec 026
|
|
145
153
|
// AC-3). The remaining samples are not taken.
|
|
146
|
-
// `mean` ± `stddev` is the per-case
|
|
154
|
+
// `mean` ± `stddev` is the per-case band, a descriptive spread (one sample sd of the N scores)
|
|
147
155
|
// used by the borderline-outcome rule and per-case drift band-overlap logic.
|
|
148
156
|
// NO DEFAULT TIMEOUT HERE (spec 017 AC-2). This defaulted to 120000, which
|
|
149
157
|
// outranked the per-surface policy exactly as lib/run.js's literal did — so the
|
package/lib/receipt.js
CHANGED
|
@@ -25,7 +25,11 @@ const SCHEMA_FILES = {
|
|
|
25
25
|
// the validator (it requires the receipt to say what answered it), so the
|
|
26
26
|
// number has to keep resolving to the schema it meant.
|
|
27
27
|
'0.5': 'receipt.v0.5.schema.json',
|
|
28
|
-
|
|
28
|
+
// v0.6 moved the same way when v0.7 took the current pointer (spec 035): v0.7 adds a
|
|
29
|
+
// verdict token a reader derives (UNDERPOWERED), and a v0.6 receipt restamped 0.7
|
|
30
|
+
// is refused by the const, so the number keeps resolving to the schema it meant.
|
|
31
|
+
'0.6': 'receipt.v0.6.schema.json',
|
|
32
|
+
'0.7': 'receipt.schema.json',
|
|
29
33
|
};
|
|
30
34
|
|
|
31
35
|
// THE BAND RULE, stated on every receipt (results.aggregates.band_rule, spec
|
|
@@ -114,7 +118,7 @@ function comparisonOf(aggWith, aggBase) {
|
|
|
114
118
|
baseline_score: aggBase.mean_score,
|
|
115
119
|
delta: noCases ? null : round(aggWith.mean_score - aggBase.mean_score),
|
|
116
120
|
// Combined uncertainty of the delta: quadrature sum of the two aggregate
|
|
117
|
-
// bands.
|
|
121
|
+
// bands. A figure beside the delta; the per-case band rule decides the headline.
|
|
118
122
|
delta_uncertainty: noCases ? null : combineUncertainty(aggWith.stddev, aggBase.stddev),
|
|
119
123
|
};
|
|
120
124
|
if (noCases) cmp.delta_uncertainty_unavailable = 'no_cases';
|
package/lib/revision.js
CHANGED
|
@@ -21,29 +21,38 @@ const { EFFECT_FLOOR } = require('../config');
|
|
|
21
21
|
|
|
22
22
|
// ── the cell headline ────────────────────────────────────────────────────────
|
|
23
23
|
// A summary of the per-case band-overlap verdicts, worded about the REVISION.
|
|
24
|
-
// The release-drift headline
|
|
25
|
-
//
|
|
24
|
+
// The release-drift headline is a sentence about a skill under a moving model.
|
|
25
|
+
// Here the model is the control. Worded under spec 031 A-031-20: a separation is
|
|
26
|
+
// detected under the rule, never proof; none detected is not evidence of sameness.
|
|
26
27
|
function revisionHeadline(perCase) {
|
|
27
28
|
const reg = perCase.filter((r) => r.verdict === 'regression').length;
|
|
28
29
|
const imp = perCase.filter((r) => r.verdict === 'improvement').length;
|
|
29
30
|
const s = (n) => (n === 1 ? '' : 's');
|
|
30
31
|
if (reg && imp) {
|
|
31
|
-
return `MIXED
|
|
32
|
+
return `MIXED: separation detected under the rule on ${imp} case${s(imp)} upward and ${reg} downward under the current upstream text (bands do not overlap).`;
|
|
32
33
|
}
|
|
33
34
|
if (reg) {
|
|
34
|
-
return `REVISION REGRESSED
|
|
35
|
+
return `REVISION REGRESSED: separation detected under the rule on ${reg} case${s(reg)}, lower under the current upstream text (bands do not overlap).`;
|
|
35
36
|
}
|
|
36
37
|
if (imp) {
|
|
37
|
-
return `REVISION IMPROVED
|
|
38
|
+
return `REVISION IMPROVED: separation detected under the rule on ${imp} case${s(imp)}, higher under the current upstream text (bands do not overlap); none lower.`;
|
|
38
39
|
}
|
|
39
|
-
return '
|
|
40
|
+
return 'NO SEPARATION DETECTED: no case separated under the rule at this sample size (band = mean \u00b1 1 sd); this is not evidence that the pinned text and the current text behave alike.';
|
|
40
41
|
}
|
|
41
42
|
|
|
42
|
-
// Classification
|
|
43
|
-
//
|
|
43
|
+
// Classification value for a cell. Kept separate so a caller can branch on the
|
|
44
|
+
// class without parsing prose. It used to be the headline's first word, cut at
|
|
45
|
+
// its dash; the headline's wording is now a display matter (spec 031 A-031-20),
|
|
46
|
+
// so the values are stated here and are unchanged: scripts/prepare-report-006.js
|
|
47
|
+
// and spec 009's gate read them. CLASS_NOT_SEPARATED is the value, not a label.
|
|
48
|
+
const CLASS_NOT_SEPARATED = 'WITHIN NOISE';
|
|
44
49
|
function revisionClass(perCase) {
|
|
45
|
-
const
|
|
46
|
-
|
|
50
|
+
const reg = perCase.filter((r) => r.verdict === 'regression').length;
|
|
51
|
+
const imp = perCase.filter((r) => r.verdict === 'improvement').length;
|
|
52
|
+
if (reg && imp) return 'MIXED';
|
|
53
|
+
if (reg) return 'REVISION REGRESSED';
|
|
54
|
+
if (imp) return 'REVISION IMPROVED';
|
|
55
|
+
return CLASS_NOT_SEPARATED;
|
|
47
56
|
}
|
|
48
57
|
|
|
49
58
|
// ── the fairness sentence ────────────────────────────────────────────────────
|
|
@@ -54,7 +63,7 @@ function revisionClass(perCase) {
|
|
|
54
63
|
// one-directional courtesy — the symmetry is a property of the code, and the
|
|
55
64
|
// gate asserts it.
|
|
56
65
|
//
|
|
57
|
-
// A cell
|
|
66
|
+
// A cell with no separation detected gets NO sentence. #005's figure stands unamended, because
|
|
58
67
|
// nothing was measured that would amend it, and manufacturing a hedge for a null
|
|
59
68
|
// result is how a report launders noise into a finding.
|
|
60
69
|
function fairnessSentence({ slug, classification, report005Delta, measuredDelta }) {
|
package/lib/stats.js
CHANGED
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
-
// Small statistics helpers for sampled
|
|
5
|
-
//
|
|
4
|
+
// Small statistics helpers for sampled scores and their bands. A band is a
|
|
5
|
+
// descriptive spread, the mean plus or minus one sample standard deviation; it
|
|
6
|
+
// carries no coverage probability. Kept dependency-free and deterministic.
|
|
6
7
|
|
|
7
8
|
function round(n, dp = 6) { const f = Math.pow(10, dp); return Math.round(n * f) / f; }
|
|
8
9
|
|
|
@@ -60,16 +61,24 @@ function aggregateBands(cases) {
|
|
|
60
61
|
return { mean: mean(means), stddev: means.length < 2 ? null : stddev(means) };
|
|
61
62
|
}
|
|
62
63
|
|
|
63
|
-
//
|
|
64
|
-
//
|
|
65
|
-
//
|
|
66
|
-
//
|
|
64
|
+
// THE NOT-SEPARATED VERDICT VALUE, which is a value and not a label (spec 031
|
|
65
|
+
// A-031-20). Callers compare against it, and every display label is mapped from
|
|
66
|
+
// it where it is rendered: lib/diff.js prints "no separation detected". The value
|
|
67
|
+
// itself is unchanged pending the verdict-rule spec.
|
|
68
|
+
const WITHIN_NOISE = 'within noise';
|
|
69
|
+
|
|
70
|
+
// Do two bands (mean ± half-width, the half-width one sample standard deviation)
|
|
71
|
+
// fail to overlap, and in which direction? Returns 'regression' (b below a),
|
|
72
|
+
// 'improvement' (b above a), or WITHIN_NOISE (the bands touch or overlap). This
|
|
73
|
+
// is the anti-false-positive rule: a separation is detected only when the bands
|
|
74
|
+
// are fully apart, and overlap is the absence of a detected separation at the
|
|
75
|
+
// sample size used, not evidence that nothing changed.
|
|
67
76
|
// regression : meanB + hwB < meanA - hwA
|
|
68
77
|
// improvement : meanB - hwB > meanA + hwA
|
|
69
78
|
function bandVerdict(meanA, hwA, meanB, hwB) {
|
|
70
79
|
if (meanB + hwB < meanA - hwA) return 'regression';
|
|
71
80
|
if (meanB - hwB > meanA + hwA) return 'improvement';
|
|
72
|
-
return
|
|
81
|
+
return WITHIN_NOISE;
|
|
73
82
|
}
|
|
74
83
|
|
|
75
84
|
// The variance ratio: how much larger the GENERATION-level spread is than the
|
|
@@ -87,4 +96,4 @@ function varianceRatio(generationSd, judgeSdMean) {
|
|
|
87
96
|
return round(g / j);
|
|
88
97
|
}
|
|
89
98
|
|
|
90
|
-
module.exports = { mean, stddev, stderr, combineUncertainty, aggregateBands, bandVerdict, round, varianceRatio };
|
|
99
|
+
module.exports = { mean, stddev, stderr, combineUncertainty, aggregateBands, bandVerdict, round, varianceRatio, WITHIN_NOISE };
|
package/lib/value.js
CHANGED
|
@@ -25,8 +25,8 @@ const { EFFECT_FLOOR } = require('../config');
|
|
|
25
25
|
//
|
|
26
26
|
// RATIO FRAMINGS ARE FLOOR-GATED. A ratio like "dollars per 0.01 lift" is
|
|
27
27
|
// only meaningful when the benefit it prices is a real move. When the lift did not clear
|
|
28
|
-
// the effect floor (or its bands overlap), the ratio renders "n/a (
|
|
29
|
-
//
|
|
28
|
+
// the effect floor (or its bands overlap), the ratio renders "n/a (no separation
|
|
29
|
+
// detected)", never a number, however tempting the arithmetic. Dividing noise by a cost
|
|
30
30
|
// produces a precise-looking figure with no evidence under it.
|
|
31
31
|
//
|
|
32
32
|
// PRICING IS FROZEN AT RUN TIME. Registry prices change (vendors cut prices; we
|
|
@@ -47,14 +47,14 @@ const COST_BASIS = {
|
|
|
47
47
|
const LATENCY_DISCLOSURE =
|
|
48
48
|
'observed on subscription CLI surface, indicative';
|
|
49
49
|
|
|
50
|
-
const NOISE_CELL = 'n/a (
|
|
50
|
+
const NOISE_CELL = 'n/a (no separation detected)';
|
|
51
51
|
|
|
52
52
|
// A floor-clearing NEGATIVE lift: the skill measurably hurt, so there is no
|
|
53
53
|
// benefit to put a price on.
|
|
54
54
|
const REGRESSED_CELL = 'n/a (skill regressed)';
|
|
55
55
|
|
|
56
56
|
// A cell with at least one separated, floor-clearing DRIVER whose AGGREGATE lift
|
|
57
|
-
// does not clear the floor. It is not
|
|
57
|
+
// does not clear the floor. It is not the no-separation cell: the QA re-derivation found
|
|
58
58
|
// six such cells rendering the noise string while the Verdict basis on the same
|
|
59
59
|
// page named their drivers (QA V-4, 2026-08-19). The aggregate stays the
|
|
60
60
|
// denominator (spec 002 AC-4, DECISIONS #9), so no price is quoted, but the cell
|
|
@@ -461,7 +461,7 @@ function costPerLiftPoint({ lift, separated, incrementalCostPer1kCalls }) {
|
|
|
461
461
|
// Two different absences, two different strings. `separated` means at least
|
|
462
462
|
// one case cleared the floor with non-overlapping bands; when that holds and
|
|
463
463
|
// only the aggregate falls short, the evidence exists and is listed under
|
|
464
|
-
// Verdict basis
|
|
464
|
+
// Verdict basis, and the no-separation cell there would contradict the same page.
|
|
465
465
|
return separated ? DRIVER_ONLY_CELL : NOISE_CELL;
|
|
466
466
|
}
|
|
467
467
|
const cost = Number(incrementalCostPer1kCalls);
|