@skillstate/bench 3.0.0 → 3.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +1 -2
  2. package/dist/ab/engagement.d.ts +70 -0
  3. package/dist/ab/engagement.d.ts.map +1 -0
  4. package/dist/ab/engagement.js +102 -0
  5. package/dist/ab/engagement.js.map +1 -0
  6. package/dist/ab/index.d.ts +28 -0
  7. package/dist/ab/index.d.ts.map +1 -0
  8. package/dist/ab/index.js +20 -0
  9. package/dist/ab/index.js.map +1 -0
  10. package/dist/ab/opencode-usage.d.ts +52 -0
  11. package/dist/ab/opencode-usage.d.ts.map +1 -0
  12. package/dist/ab/opencode-usage.js +85 -0
  13. package/dist/ab/opencode-usage.js.map +1 -0
  14. package/dist/ab/record.d.ts +173 -0
  15. package/dist/ab/record.d.ts.map +1 -0
  16. package/dist/ab/record.js +98 -0
  17. package/dist/ab/record.js.map +1 -0
  18. package/dist/ab/replay.d.ts +193 -0
  19. package/dist/ab/replay.d.ts.map +1 -0
  20. package/dist/ab/replay.js +149 -0
  21. package/dist/ab/replay.js.map +1 -0
  22. package/dist/ab/report.d.ts +26 -0
  23. package/dist/ab/report.d.ts.map +1 -0
  24. package/dist/ab/report.js +98 -0
  25. package/dist/ab/report.js.map +1 -0
  26. package/dist/ab/stats-core.d.ts +84 -0
  27. package/dist/ab/stats-core.d.ts.map +1 -0
  28. package/dist/ab/stats-core.js +93 -0
  29. package/dist/ab/stats-core.js.map +1 -0
  30. package/dist/ab/survey.d.ts +131 -0
  31. package/dist/ab/survey.d.ts.map +1 -0
  32. package/dist/ab/survey.js +143 -0
  33. package/dist/ab/survey.js.map +1 -0
  34. package/dist/ab/usage.d.ts +99 -0
  35. package/dist/ab/usage.d.ts.map +1 -0
  36. package/dist/ab/usage.js +98 -0
  37. package/dist/ab/usage.js.map +1 -0
  38. package/dist/ab/verdict.d.ts +88 -0
  39. package/dist/ab/verdict.d.ts.map +1 -0
  40. package/dist/ab/verdict.js +251 -0
  41. package/dist/ab/verdict.js.map +1 -0
  42. package/dist/ab-cli.d.ts +35 -0
  43. package/dist/ab-cli.d.ts.map +1 -0
  44. package/dist/ab-cli.js +161 -0
  45. package/dist/ab-cli.js.map +1 -0
  46. package/dist/index.d.ts +1 -0
  47. package/dist/index.d.ts.map +1 -1
  48. package/dist/index.js +1 -0
  49. package/dist/index.js.map +1 -1
  50. package/package.json +5 -1
@@ -0,0 +1,98 @@
1
+ /**
2
+ * Token accounting read from the HOST's own store.
3
+ *
4
+ * ── Why not parse the host's stdout ──────────────────────────────────────
5
+ *
6
+ * Three reasons, in order of how much they matter:
7
+ *
8
+ * 1. **The model does not report its own cost reliably.** A harness that
9
+ * takes the number the subject of the experiment wrote down is trusting the
10
+ * measurement to the thing being measured. The previous A/B took its
11
+ * figures from the run's own reporting, and one of its two "arms" was
12
+ * itself two different readings of a single session.
13
+ * 2. **OpenCode already records per-message token usage** in its local store
14
+ * (`message.data.tokens` with `input`, `output`, `reasoning` and
15
+ * `cache.{read,write}`). That is the host's own accounting, not a
16
+ * reconstruction, and it is keyed by session so a run can be re-read
17
+ * after the fact.
18
+ * 3. **It is available without a live model.** Reading the store works when
19
+ * the provider quota is exhausted, which is exactly when one most wants to
20
+ * re-analyse a previous run.
21
+ *
22
+ * Node's `node:sqlite` is used rather than a dependency: the store is a
23
+ * local file, the read is a single prepared statement, and adding a native
24
+ * dependency to a zero-deps package for this would be a poor trade. The
25
+ * reader is injected, so the query and its failure modes are testable
26
+ * without an OpenCode installation.
27
+ *
28
+ * @non-paper — measurement infrastructure, not the paper's evaluation.
29
+ */
30
+ /** Zero usage, for a session whose messages carry no accounting. */
31
+ export const NO_USAGE = {
32
+ input: 0,
33
+ cacheRead: 0,
34
+ cacheWrite: 0,
35
+ output: 0,
36
+ };
37
+ /** Sum one row's token fields, tolerating a partial or absent record. */
38
+ function rowUsage(tokens) {
39
+ return {
40
+ input: tokens?.input ?? 0,
41
+ cacheRead: tokens?.cacheRead ?? 0,
42
+ cacheWrite: tokens?.cacheWrite ?? 0,
43
+ output: tokens?.output ?? 0,
44
+ };
45
+ }
46
+ /**
47
+ * Total token spend for a session.
48
+ *
49
+ * Only `assistant` rows count. A user message's `input` figure is the host
50
+ * describing the prompt it echoed, and adding it would double-count the
51
+ * conversation.
52
+ *
53
+ * `cacheRead` is summed, not treated as free: prompt caching changes the
54
+ * price of a token, not the fact that the model was shown it, and the
55
+ * paper's claim is about what the model is exposed to.
56
+ */
57
+ export function sessionUsage(rows) {
58
+ return rows
59
+ .filter((row) => row.role === 'assistant')
60
+ .map((row) => rowUsage(row.tokens))
61
+ .reduce((sum, usage) => ({
62
+ input: sum.input + usage.input,
63
+ cacheRead: sum.cacheRead + usage.cacheRead,
64
+ cacheWrite: sum.cacheWrite + usage.cacheWrite,
65
+ output: sum.output + usage.output,
66
+ }), NO_USAGE);
67
+ }
68
+ /**
69
+ * Resolve a session's token spend, turning every failure into a value.
70
+ *
71
+ * A run whose accounting cannot be read is a run with unknown cost, which is
72
+ * not the same as a run that cost nothing. The distinction is the whole
73
+ * reason this returns a union instead of a number: `store_unavailable` must
74
+ * never be silently read as `NO_USAGE`.
75
+ */
76
+ export async function resolveSessionUsage(reader, sessionID) {
77
+ let rows;
78
+ try {
79
+ rows = await reader.messagesFor(sessionID);
80
+ }
81
+ catch (error) {
82
+ return {
83
+ ok: false,
84
+ reason: 'store_unavailable',
85
+ detail: error instanceof Error ? error.message : String(error),
86
+ };
87
+ }
88
+ if (rows.length === 0) {
89
+ return {
90
+ ok: false,
91
+ reason: 'session_unknown',
92
+ detail: `no messages recorded for session ${sessionID}`,
93
+ };
94
+ }
95
+ const assistant = rows.filter((row) => row.role === 'assistant').length;
96
+ return { ok: true, usage: sessionUsage(rows), assistantMessages: assistant };
97
+ }
98
+ //# sourceMappingURL=usage.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"usage.js","sourceRoot":"","sources":["../../src/ab/usage.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA4BG;AAmCH,oEAAoE;AACpE,MAAM,CAAC,MAAM,QAAQ,GAAe;IAClC,KAAK,EAAE,CAAC;IACR,SAAS,EAAE,CAAC;IACZ,UAAU,EAAE,CAAC;IACb,MAAM,EAAE,CAAC;CACV,CAAC;AAEF,yEAAyE;AACzE,SAAS,QAAQ,CAAC,MAAkC;IAClD,OAAO;QACL,KAAK,EAAE,MAAM,EAAE,KAAK,IAAI,CAAC;QACzB,SAAS,EAAE,MAAM,EAAE,SAAS,IAAI,CAAC;QACjC,UAAU,EAAE,MAAM,EAAE,UAAU,IAAI,CAAC;QACnC,MAAM,EAAE,MAAM,EAAE,MAAM,IAAI,CAAC;KAC5B,CAAC;AACJ,CAAC;AAED;;;;;;;;;;GAUG;AACH,MAAM,UAAU,YAAY,CAAC,IAA+B;IAC1D,OAAO,IAAI;SACR,MAAM,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,IAAI,KAAK,WAAW,CAAC;SACzC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,QAAQ,CAAC,GAAG,CAAC,MAAM,CAAC,CAAC;SAClC,MAAM,CACL,CAAC,GAAG,EAAE,KAAK,EAAE,EAAE,CAAC,CAAC;QACf,KAAK,EAAE,GAAG,CAAC,KAAK,GAAG,KAAK,CAAC,KAAK;QAC9B,SAAS,EAAE,GAAG,CAAC,SAAS,GAAG,KAAK,CAAC,SAAS;QAC1C,UAAU,EAAE,GAAG,CAAC,UAAU,GAAG,KAAK,CAAC,UAAU;QAC7C,MAAM,EAAE,GAAG,CAAC,MAAM,GAAG,KAAK,CAAC,MAAM;KAClC,CAAC,EACF,QAAQ,CACT,CAAC;AACN,CAAC;AAcD;;;;;;;GAOG;AACH,MAAM,CAAC,KAAK,UAAU,mBAAmB,CACvC,MAAmB,EACnB,SAAiB;IAEjB,IAAI,IAA+B,CAAC;IACpC,IAAI,CAAC;QACH,IAAI,GAAG,MAAM,MAAM,CAAC,WAAW,CAAC,SAAS,CAAC,CAAC;IAC7C,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,OAAO;YACL,EAAE,EAAE,KAAK;YACT,MAAM,EAAE,mBAAmB;YAC3B,MAAM,EAAE,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC;SAC/D,CAAC;IACJ,CAAC;IACD,IAAI,IAAI,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACtB,OAAO;YACL,EAAE,EAAE,KAAK;YACT,MAAM,EAAE,iBAAiB;YACzB,MAAM,EAAE,oCAAoC,SAAS,EAAE;SACxD,CAAC;IACJ,CAAC;IACD,MAAM,SAAS,GAAG,IAAI,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,IAAI,KAAK,WAAW,CAAC,CAAC,MAAM,CAAC;IACxE,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,KAAK,EAAE,YAAY,CAAC,IAAI,CAAC,EAAE,iBAAiB,EAAE,SAAS,EAAE,CAAC;AAC/E,CAAC"}
@@ -0,0 +1,88 @@
1
+ /**
2
+ * The gates: what has to be true before a token comparison may be reported.
3
+ *
4
+ * ── What went wrong the first time ───────────────────────────────────────
5
+ *
6
+ * The previous A/B reported a 39% saving from a run in which the
7
+ * instrumented arm never once wrote the state file. Nothing flagged it,
8
+ * because the pipeline carried a number and a number has no opinion about
9
+ * whether the thing it measured was switched on.
10
+ *
11
+ * So the harness's primary output is a {@link Verdict}, and a bare percentage
12
+ * is only reachable by passing every gate. The gates are deliberately
13
+ * pessimistic and deliberately ordered: the most invalidating finding is
14
+ * computed first, so a later gate cannot refine a comparison that was never
15
+ * a comparison.
16
+ *
17
+ * @non-paper — measurement infrastructure for OUR host integration, not the
18
+ * paper's own evaluation. The paper's Limitations are cited here as the
19
+ * source of one gate, not as a claim being verified.
20
+ */
21
+ import type { ArmId, ArmRecord } from './record.js';
22
+ import type { Distribution, EffectSize } from './stats-core.js';
23
+ /** Minimum trials per arm before a comparison may claim anything. */
24
+ export declare const MIN_TRIALS = 2;
25
+ /**
26
+ * What the harness concluded.
27
+ *
28
+ * The last four are refusals, and refusals are the point: a harness that can
29
+ * only produce numbers is a harness that will produce the wrong number.
30
+ */
31
+ export type Verdict =
32
+ /** Engaged, comparable, and the instrumented arm spent significantly fewer prompt tokens. */
33
+ 'saving'
34
+ /** Engaged, comparable, and the instrumented arm spent significantly more. */
35
+ | 'regression'
36
+ /** A valid run whose effect sits inside the noise. */
37
+ | 'no-effect'
38
+ /** The instrumented arm never wrote state. No number is reported. */
39
+ | 'inert'
40
+ /** The arms did not do comparable work, so their tokens are not comparable. */
41
+ | 'not-comparable'
42
+ /** A failed run, a mismatched task/model/host, or too few trials. */
43
+ | 'invalid';
44
+ /** A single gate that refused to let a number be reported. */
45
+ export interface GateFailure {
46
+ readonly gate: 'comparability' | 'completeness' | 'sample-size' | 'engagement' | 'task-equivalence' | 'outcome' | 'variance' | 'paper-compatibility';
47
+ /** Why the gate fired, phrased so a reader can act on it. */
48
+ readonly detail: string;
49
+ }
50
+ export interface VerdictResult {
51
+ readonly verdict: Verdict;
52
+ /** Every gate that fired, in gate order. */
53
+ readonly failures: readonly GateFailure[];
54
+ /** Per-arm prompt-token distributions; empty arms carry `n: 0`. */
55
+ readonly distributions: Readonly<Record<ArmId, Distribution>>;
56
+ /** The effect size, only when both arms have usable samples. */
57
+ readonly effect: EffectSize | null;
58
+ /** One line for printing. Never a bare percentage on a refusal. */
59
+ readonly summary: string;
60
+ }
61
+ /** Options for {@link runExperiment}. */
62
+ export interface ExperimentOptions {
63
+ /**
64
+ * Set when the task's objective is defined over the historical trajectory
65
+ * (an audit, a review, a "what did I just do" task).
66
+ *
67
+ * Limitations, case (3): the paper's own assumption fails exactly there. The
68
+ * section is not numbered — `state.md` records it as `§ Limitations` with no
69
+ * number, so a number here would be a citation nobody can check. An
70
+ * unverifiable citation is worse than none: it looks verified.
71
+ * A flat result on such a task is the predicted outcome, so reporting it
72
+ * as a refutation would be a category error in the harness's own favour.
73
+ * The gate downgrades `no-effect` to a failure that says so.
74
+ */
75
+ readonly taskNeedsTranscript?: boolean;
76
+ /** Trials required per arm. Defaults to {@link MIN_TRIALS}. */
77
+ readonly minTrials?: number;
78
+ }
79
+ /**
80
+ * Run every gate and return the strongest conclusion the data supports.
81
+ *
82
+ * Pure and deterministic: no clock, no filesystem, no randomness. Identical
83
+ * input yields an identical result, which is what lets the gates be tested
84
+ * against the real historical runs and against inputs built to trip each
85
+ * branch.
86
+ */
87
+ export declare function runExperiment(arms: ReadonlyMap<ArmId, ArmRecord>, options?: ExperimentOptions): VerdictResult;
88
+ //# sourceMappingURL=verdict.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"verdict.d.ts","sourceRoot":"","sources":["../../src/ab/verdict.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAGH,OAAO,KAAK,EAAE,KAAK,EAAE,SAAS,EAAE,MAAM,aAAa,CAAC;AAEpD,OAAO,KAAK,EAAE,YAAY,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAEhE,qEAAqE;AACrE,eAAO,MAAM,UAAU,IAAI,CAAC;AAE5B;;;;;GAKG;AACH,MAAM,MAAM,OAAO;AACjB,6FAA6F;AAC3F,QAAQ;AACV,8EAA8E;GAC5E,YAAY;AACd,sDAAsD;GACpD,WAAW;AACb,qEAAqE;GACnE,OAAO;AACT,+EAA+E;GAC7E,gBAAgB;AAClB,qEAAqE;GACnE,SAAS,CAAC;AAEd,8DAA8D;AAC9D,MAAM,WAAW,WAAW;IAC1B,QAAQ,CAAC,IAAI,EACT,eAAe,GACf,cAAc,GACd,aAAa,GACb,YAAY,GACZ,kBAAkB,GAClB,SAAS,GACT,UAAU,GACV,qBAAqB,CAAC;IAC1B,6DAA6D;IAC7D,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;CACzB;AAED,MAAM,WAAW,aAAa;IAC5B,QAAQ,CAAC,OAAO,EAAE,OAAO,CAAC;IAC1B,4CAA4C;IAC5C,QAAQ,CAAC,QAAQ,EAAE,SAAS,WAAW,EAAE,CAAC;IAC1C,mEAAmE;IACnE,QAAQ,CAAC,aAAa,EAAE,QAAQ,CAAC,MAAM,CAAC,KAAK,EAAE,YAAY,CAAC,CAAC,CAAC;IAC9D,gEAAgE;IAChE,QAAQ,CAAC,MAAM,EAAE,UAAU,GAAG,IAAI,CAAC;IACnC,mEAAmE;IACnE,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;CAC1B;AAED,yCAAyC;AACzC,MAAM,WAAW,iBAAiB;IAChC;;;;;;;;;;;OAWG;IACH,QAAQ,CAAC,mBAAmB,CAAC,EAAE,OAAO,CAAC;IACvC,+DAA+D;IAC/D,QAAQ,CAAC,SAAS,CAAC,EAAE,MAAM,CAAC;CAC7B;AA0BD;;;;;;;GAOG;AACH,wBAAgB,aAAa,CAC3B,IAAI,EAAE,WAAW,CAAC,KAAK,EAAE,SAAS,CAAC,EACnC,OAAO,GAAE,iBAAsB,GAC9B,aAAa,CAiOf"}
@@ -0,0 +1,251 @@
1
+ /**
2
+ * The gates: what has to be true before a token comparison may be reported.
3
+ *
4
+ * ── What went wrong the first time ───────────────────────────────────────
5
+ *
6
+ * The previous A/B reported a 39% saving from a run in which the
7
+ * instrumented arm never once wrote the state file. Nothing flagged it,
8
+ * because the pipeline carried a number and a number has no opinion about
9
+ * whether the thing it measured was switched on.
10
+ *
11
+ * So the harness's primary output is a {@link Verdict}, and a bare percentage
12
+ * is only reachable by passing every gate. The gates are deliberately
13
+ * pessimistic and deliberately ordered: the most invalidating finding is
14
+ * computed first, so a later gate cannot refine a comparison that was never
15
+ * a comparison.
16
+ *
17
+ * @non-paper — measurement infrastructure for OUR host integration, not the
18
+ * paper's own evaluation. The paper's Limitations are cited here as the
19
+ * source of one gate, not as a claim being verified.
20
+ */
21
+ import { isInstrumented, promptTokens } from './record.js';
22
+ import { describe, effectSize, MIN_EFFECT_IN_MADS } from './stats-core.js';
23
+ /** Minimum trials per arm before a comparison may claim anything. */
24
+ export const MIN_TRIALS = 2;
25
+ const EMPTY_DISTRIBUTION = {
26
+ n: 0,
27
+ min: 0,
28
+ median: 0,
29
+ max: 0,
30
+ mad: 0,
31
+ relativeMad: 0,
32
+ sorted: [],
33
+ };
34
+ function distributionMap(arms) {
35
+ const out = {
36
+ plain: EMPTY_DISTRIBUTION,
37
+ notes: EMPTY_DISTRIBUTION,
38
+ paper: EMPTY_DISTRIBUTION,
39
+ };
40
+ for (const [arm, record] of arms) {
41
+ out[arm] = describe(record.runs.map((run) => promptTokens(run.usage)));
42
+ }
43
+ return out;
44
+ }
45
+ /**
46
+ * Run every gate and return the strongest conclusion the data supports.
47
+ *
48
+ * Pure and deterministic: no clock, no filesystem, no randomness. Identical
49
+ * input yields an identical result, which is what lets the gates be tested
50
+ * against the real historical runs and against inputs built to trip each
51
+ * branch.
52
+ */
53
+ export function runExperiment(arms, options = {}) {
54
+ const minTrials = options.minTrials ?? MIN_TRIALS;
55
+ const failures = [];
56
+ const distributions = distributionMap(arms);
57
+ const control = arms.get('plain');
58
+ const instrumented = [...arms.entries()].filter(([arm]) => isInstrumented(arm));
59
+ // Every gate is evaluated before any early return, so a caller fixing a
60
+ // broken experiment is told everything that is wrong in one pass rather
61
+ // than discovering the next problem on the next run. Only gates that are
62
+ // undefined without a control arm (equivalence, effect) are skipped.
63
+ const hasBothArms = control !== undefined && instrumented.length > 0;
64
+ if (!hasBothArms) {
65
+ failures.push({
66
+ gate: 'comparability',
67
+ detail: 'need a `plain` control arm and at least one instrumented arm',
68
+ });
69
+ }
70
+ // ── Gate 1: comparability — same task, same model, same host ───────────
71
+ for (const [arm, record] of hasBothArms ? instrumented : []) {
72
+ const baseline = control;
73
+ if (record.task !== baseline.task ||
74
+ record.model !== baseline.model ||
75
+ record.hostVersion !== baseline.hostVersion) {
76
+ failures.push({
77
+ gate: 'comparability',
78
+ detail: `arm ${arm} ran ${JSON.stringify(record.task)} on ${record.model}@${record.hostVersion}` +
79
+ ` but the control ran ${JSON.stringify(baseline.task)} on ${baseline.model}@${baseline.hostVersion}`,
80
+ });
81
+ }
82
+ }
83
+ // ── Gate 2: completeness — every run actually finished ────────────────
84
+ for (const [arm, record] of arms) {
85
+ for (const run of record.runs) {
86
+ if (!run.completed) {
87
+ failures.push({
88
+ gate: 'completeness',
89
+ detail: `arm ${arm} trial ${run.trial} did not complete: ${run.error ?? 'unknown error'}`,
90
+ });
91
+ }
92
+ }
93
+ }
94
+ // ── Gate 2b: the outcome — did the work get done? ───────────────────────
95
+ //
96
+ // The last gate, and the one that was missing. Every other gate asks whether
97
+ // the comparison is valid; none asks whether the result is worth anything.
98
+ // An arm that reads every file, writes state and never produces the answer
99
+ // passes engagement, comparability and variance, and would be reported as a
100
+ // saving. That is the failure this project has already paid for once: "paper
101
+ // 81,685 tokens vs control 203,801 = 60% cheaper, paper answered WRONG".
102
+ //
103
+ // Two refusals, and the second is the one that matters most. An experiment
104
+ // that measured cost and not the work has not earned the right to report the
105
+ // cost, whatever its token numbers say.
106
+ const measured = (record) => record.runs.some((run) => run.outcome !== undefined);
107
+ for (const [arm, record] of instrumented) {
108
+ if (!measured(record) || control === undefined || !measured(control)) {
109
+ failures.push({
110
+ gate: 'outcome',
111
+ detail: `arm ${arm} did not record whether the task was answered — a cost ` +
112
+ 'saving with no outcome is not a result, so none is reported. Pass ' +
113
+ '`correct` in the trial files to measure it.',
114
+ });
115
+ continue;
116
+ }
117
+ const correct = (record) => record.runs.filter((run) => run.outcome?.correct === true).length;
118
+ // Strictly, not "at least as good". The n=7 experiment's honest reading
119
+ // was that the arms are comparably accurate and only the cost differs, and
120
+ // a gate that fired on a tie would have thrown that finding away.
121
+ if (correct(record) / record.runs.length < correct(control) / control.runs.length) {
122
+ failures.push({
123
+ gate: 'outcome',
124
+ detail: `arm ${arm} answered correctly ${correct(record)} of ${record.runs.length} trial(s) ` +
125
+ `while the control answered ${correct(control)} of ${control.runs.length} — ` +
126
+ 'a cheaper run that did not finish the work is a regression, not a saving',
127
+ });
128
+ }
129
+ }
130
+ // ── Gate 3: sample size — one run cannot show variance ────────────────
131
+ for (const [arm, record] of arms) {
132
+ if (record.runs.length < minTrials) {
133
+ failures.push({
134
+ gate: 'sample-size',
135
+ detail: `arm ${arm} ran ${record.runs.length} trial(s); ${minTrials} are needed` +
136
+ ' to separate an effect from run-to-run variance',
137
+ });
138
+ }
139
+ }
140
+ // ── Gate 4: engagement — was the integration switched on at all? ───────
141
+ //
142
+ // First in the list of reasons a run is unusable, because when it fires
143
+ // every token number in the experiment is a sample of model variance. The
144
+ // detail distinguishes "never engaged" from "engaged but every response
145
+ // was rejected", which are different bugs with different fixes.
146
+ for (const [arm, record] of instrumented) {
147
+ if (record.runs.every((run) => !run.engagement.engaged)) {
148
+ const rejections = record.runs.reduce((n, run) => n + run.engagement.rejections, 0);
149
+ const first = record.runs.find((run) => run.engagement.firstRejection !== undefined)?.engagement.firstRejection;
150
+ const detail = rejections > 0
151
+ ? `arm ${arm} never wrote state and rejected ${rejections} response(s)` +
152
+ (first === undefined ? '' : ` (first: ${first})`) +
153
+ ' — the model tried and the integration refused every patch, so its tokens measure the rejection, not the integration'
154
+ : `arm ${arm} never wrote the state file in ${record.runs.length} trial(s) — the integration was inert, so its token count is not a measurement of the integration`;
155
+ failures.push({ gate: 'engagement', detail });
156
+ }
157
+ }
158
+ // ── Gate 5: task equivalence — did both arms do the same work? ─────────
159
+ //
160
+ // Byte-identical artifacts are the strongest available evidence that both
161
+ // arms finished the same thing. A digest that appears in no control run
162
+ // means the arms are not comparable, whatever their token counts say.
163
+ const controlDigests = new Set(control === undefined ? [] : control.runs.map((run) => run.work.artifactDigest));
164
+ for (const [arm, record] of hasBothArms ? instrumented : []) {
165
+ for (const run of record.runs) {
166
+ if (!controlDigests.has(run.work.artifactDigest)) {
167
+ failures.push({
168
+ gate: 'task-equivalence',
169
+ detail: `arm ${arm} trial ${run.trial} produced artifact digest ${String(run.work.artifactDigest)},` +
170
+ ` which is not among the control's (${[...controlDigests].map(String).join(', ')})`,
171
+ });
172
+ }
173
+ }
174
+ }
175
+ // Gates 1, 2, 3 and 5 all invalidate the comparison itself. Engagement (4)
176
+ // gets its own verdict, because "the integration did nothing" is a
177
+ // materially different finding from "the arms did different work".
178
+ const blocking = failures;
179
+ if (blocking.length > 0) {
180
+ const inert = blocking.some((failure) => failure.gate === 'engagement');
181
+ const invalid = blocking.some((failure) => failure.gate === 'comparability' ||
182
+ failure.gate === 'completeness' ||
183
+ failure.gate === 'sample-size');
184
+ return refuse(inert ? 'inert' : invalid ? 'invalid' : 'not-comparable', blocking, distributions);
185
+ }
186
+ // ── Gate 6: variance — is the effect bigger than the noise? ────────────
187
+ const [instrumentedArm, instrumentedRecord] = instrumented[0];
188
+ const effect = effectSize(control.runs.map((run) => promptTokens(run.usage)), instrumentedRecord.runs.map((run) => promptTokens(run.usage)));
189
+ if (effect.inMads !== null && Math.abs(effect.inMads) < MIN_EFFECT_IN_MADS) {
190
+ failures.push({
191
+ gate: 'variance',
192
+ detail: `effect is ${effect.inMads.toFixed(2)} MADs, under the ${MIN_EFFECT_IN_MADS} required to call it a signal;` +
193
+ ` the ${instrumentedArm} arm's own spread is ±${(distributions[instrumentedArm].relativeMad * 100).toFixed(0)}% of its median`,
194
+ });
195
+ }
196
+ // ── Gate 7: paper compatibility — a null on the wrong task is not a null ─
197
+ if (failures.length === 0 && options.taskNeedsTranscript === true) {
198
+ failures.push({
199
+ gate: 'paper-compatibility',
200
+ detail: 'this task is defined over the historical trajectory (audit-style), which is the case the paper’s Limitations call out' +
201
+ ' predicts will not benefit from a bounded prompt; a flat result here is consistent with the paper, not a' +
202
+ ' refutation of it — measure a task that needs cross-turn memory',
203
+ });
204
+ }
205
+ if (failures.length > 0) {
206
+ // A flat result on a transcript-shaped task is not a measurement of the
207
+ // integration at all, so it gets its own headline: a reader who sees
208
+ // "NO-EFFECT" would reasonably conclude skillstate does not help, when
209
+ // the honest statement is that this task was never a test of it.
210
+ const onTranscriptTask = options.taskNeedsTranscript === true &&
211
+ failures[0].gate === 'paper-compatibility';
212
+ return {
213
+ verdict: 'no-effect',
214
+ failures,
215
+ distributions,
216
+ effect,
217
+ summary: `${onTranscriptTask ? 'NOT-A-TEST' : 'NO-EFFECT'}: ${failures[0].detail} [${failures.length} gate(s) failed]`,
218
+ };
219
+ }
220
+ const direction = effect.difference > 0 ? 'fewer' : 'more';
221
+ const magnitude = effect.inMads === null
222
+ ? 'exact: both arms had zero spread'
223
+ : `${Math.abs(effect.inMads).toFixed(1)} MADs`;
224
+ return {
225
+ verdict: effect.difference > 0 ? 'saving' : 'regression',
226
+ failures,
227
+ distributions,
228
+ effect,
229
+ summary: `${effect.difference > 0 ? 'SAVING' : 'REGRESSION'}: arm ${instrumentedArm} spent ` +
230
+ `${Math.abs(effect.difference)} ${direction} prompt tokens at the median (${magnitude})`,
231
+ };
232
+ }
233
+ /** Build a refusal result. Every gate failure is reported, never just the first. */
234
+ function refuse(verdict, failures, distributions) {
235
+ return {
236
+ verdict,
237
+ failures,
238
+ distributions,
239
+ effect: null,
240
+ summary: `${HEADLINE[verdict]}: ${failures[0].detail} [${failures.length} gate(s) failed]`,
241
+ };
242
+ }
243
+ const HEADLINE = {
244
+ saving: 'SAVING',
245
+ regression: 'REGRESSION',
246
+ 'no-effect': 'NO-EFFECT',
247
+ inert: 'INERT',
248
+ 'not-comparable': 'NOT-COMPARABLE',
249
+ invalid: 'INVALID',
250
+ };
251
+ //# sourceMappingURL=verdict.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"verdict.js","sourceRoot":"","sources":["../../src/ab/verdict.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH,OAAO,EAAE,cAAc,EAAE,YAAY,EAAE,MAAM,aAAa,CAAC;AAE3D,OAAO,EAAE,QAAQ,EAAE,UAAU,EAAE,kBAAkB,EAAE,MAAM,iBAAiB,CAAC;AAG3E,qEAAqE;AACrE,MAAM,CAAC,MAAM,UAAU,GAAG,CAAC,CAAC;AAoE5B,MAAM,kBAAkB,GAAiB;IACvC,CAAC,EAAE,CAAC;IACJ,GAAG,EAAE,CAAC;IACN,MAAM,EAAE,CAAC;IACT,GAAG,EAAE,CAAC;IACN,GAAG,EAAE,CAAC;IACN,WAAW,EAAE,CAAC;IACd,MAAM,EAAE,EAAE;CACX,CAAC;AAEF,SAAS,eAAe,CACtB,IAAmC;IAEnC,MAAM,GAAG,GAAgC;QACvC,KAAK,EAAE,kBAAkB;QACzB,KAAK,EAAE,kBAAkB;QACzB,KAAK,EAAE,kBAAkB;KAC1B,CAAC;IACF,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,IAAI,EAAE,CAAC;QACjC,GAAG,CAAC,GAAG,CAAC,GAAG,QAAQ,CAAC,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,YAAY,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC;IACzE,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,aAAa,CAC3B,IAAmC,EACnC,OAAO,GAAsB,EAAE;IAE/B,MAAM,SAAS,GAAG,OAAO,CAAC,SAAS,IAAI,UAAU,CAAC;IAClD,MAAM,QAAQ,GAAkB,EAAE,CAAC;IACnC,MAAM,aAAa,GAAG,eAAe,CAAC,IAAI,CAAC,CAAC;IAE5C,MAAM,OAAO,GAAG,IAAI,CAAC,GAAG,CAAC,OAAO,CAAC,CAAC;IAClC,MAAM,YAAY,GAAG,CAAC,GAAG,IAAI,CAAC,OAAO,EAAE,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,GAAG,CAAC,EAAE,EAAE,CAAC,cAAc,CAAC,GAAG,CAAC,CAAC,CAAC;IAEhF,wEAAwE;IACxE,wEAAwE;IACxE,yEAAyE;IACzE,qEAAqE;IACrE,MAAM,WAAW,GAAG,OAAO,KAAK,SAAS,IAAI,YAAY,CAAC,MAAM,GAAG,CAAC,CAAC;IACrE,IAAI,CAAC,WAAW,EAAE,CAAC;QACjB,QAAQ,CAAC,IAAI,CAAC;YACZ,IAAI,EAAE,eAAe;YACrB,MAAM,EAAE,8DAA8D;SACvE,CAAC,CAAC;IACL,CAAC;IAED,0EAA0E;IAC1E,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,WAAW,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC;QAC5D,MAAM,QAAQ,GAAG,OAAQ,CAAC;QAC1B,IACE,MAAM,CAAC,IAAI,KAAK,QAAQ,CAAC,IAAI;YAC7B,MAAM,CAAC,KAAK,KAAK,QAAQ,CAAC,KAAK;YAC/B,MAAM,CAAC,WAAW,KAAK,QAAQ,CAAC,WAAW,EAC3C,CAAC;YACD,QAAQ,CAAC,IAAI,CAAC;gBACZ,IAAI,EAAE,eAAe;gBACrB,MAAM,EACJ,OAAO,GAAG,QAAQ,IAAI,CAAC,SAAS,CAAC,MAAM,CAAC,IAAI,CAAC,OAAO,MAAM,CAAC,KAAK,IAAI,MAAM,CAAC,WAAW,EAAE;oBACxF,wBAAwB,IAAI,CAAC,SAAS,CAAC,QAAQ,CAAC,IAAI,CAAC,OAAO,QAAQ,CAAC,KAAK,IAAI,QAAQ,CAAC,WAAW,EAAE;aACvG,CAAC,CAAC;QACL,CAAC;IACH,CAAC;IAED,yEAAyE;IACzE,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,IAAI,EAAE,CAAC;QACjC,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,IAAI,EAAE,CAAC;YAC9B,IAAI,CAAC,GAAG,CAAC,SAAS,EAAE,CAAC;gBACnB,QAAQ,CAAC,IAAI,CAAC;oBACZ,IAAI,EAAE,cAAc;oBACpB,MAAM,EAAE,OAAO,GAAG,UAAU,GAAG,CAAC,KAAK,sBAAsB,GAAG,CAAC,KAAK,IAAI,eAAe,EAAE;iBAC1F,CAAC,CAAC;YACL,CAAC;QACH,CAAC;IACH,CAAC;IAED,2EAA2E;IAC3E,EAAE;IACF,6EAA6E;IAC7E,2EAA2E;IAC3E,2EAA2E;IAC3E,4EAA4E;IAC5E,6EAA6E;IAC7E,yEAAyE;IACzE,EAAE;IACF,2EAA2E;IAC3E,6EAA6E;IAC7E,wCAAwC;IACxC,MAAM,QAAQ,GAAG,CAAC,MAAiB,EAAW,EAAE,CAC9C,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,OAAO,KAAK,SAAS,CAAC,CAAC;IACvD,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,YAAY,EAAE,CAAC;QACzC,IAAI,CAAC,QAAQ,CAAC,MAAM,CAAC,IAAI,OAAO,KAAK,SAAS,IAAI,CAAC,QAAQ,CAAC,OAAO,CAAC,EAAE,CAAC;YACrE,QAAQ,CAAC,IAAI,CAAC;gBACZ,IAAI,EAAE,SAAS;gBACf,MAAM,EACJ,OAAO,GAAG,yDAAyD;oBACnE,oEAAoE;oBACpE,6CAA6C;aAChD,CAAC,CAAC;YACH,SAAS;QACX,CAAC;QACD,MAAM,OAAO,GAAG,CAAC,MAAiB,EAAU,EAAE,CAC5C,MAAM,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,OAAO,EAAE,OAAO,KAAK,IAAI,CAAC,CAAC,MAAM,CAAC;QACpE,wEAAwE;QACxE,2EAA2E;QAC3E,kEAAkE;QAClE,IAAI,OAAO,CAAC,MAAM,CAAC,GAAG,MAAM,CAAC,IAAI,CAAC,MAAM,GAAG,OAAO,CAAC,OAAO,CAAC,GAAG,OAAO,CAAC,IAAI,CAAC,MAAM,EAAE,CAAC;YAClF,QAAQ,CAAC,IAAI,CAAC;gBACZ,IAAI,EAAE,SAAS;gBACf,MAAM,EACJ,OAAO,GAAG,uBAAuB,OAAO,CAAC,MAAM,CAAC,OAAO,MAAM,CAAC,IAAI,CAAC,MAAM,YAAY;oBACrF,8BAA8B,OAAO,CAAC,OAAO,CAAC,OAAO,OAAO,CAAC,IAAI,CAAC,MAAM,KAAK;oBAC7E,0EAA0E;aAC7E,CAAC,CAAC;QACL,CAAC;IACH,CAAC;IAED,yEAAyE;IACzE,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,IAAI,EAAE,CAAC;QACjC,IAAI,MAAM,CAAC,IAAI,CAAC,MAAM,GAAG,SAAS,EAAE,CAAC;YACnC,QAAQ,CAAC,IAAI,CAAC;gBACZ,IAAI,EAAE,aAAa;gBACnB,MAAM,EACJ,OAAO,GAAG,QAAQ,MAAM,CAAC,IAAI,CAAC,MAAM,cAAc,SAAS,aAAa;oBACxE,iDAAiD;aACpD,CAAC,CAAC;QACL,CAAC;IACH,CAAC;IAED,0EAA0E;IAC1E,EAAE;IACF,wEAAwE;IACxE,0EAA0E;IAC1E,wEAAwE;IACxE,gEAAgE;IAChE,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,YAAY,EAAE,CAAC;QACzC,IAAI,MAAM,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,UAAU,CAAC,OAAO,CAAC,EAAE,CAAC;YACxD,MAAM,UAAU,GAAG,MAAM,CAAC,IAAI,CAAC,MAAM,CACnC,CAAC,CAAC,EAAE,GAAG,EAAE,EAAE,CAAC,CAAC,GAAG,GAAG,CAAC,UAAU,CAAC,UAAU,EACzC,CAAC,CACF,CAAC;YACF,MAAM,KAAK,GAAG,MAAM,CAAC,IAAI,CAAC,IAAI,CAC5B,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,UAAU,CAAC,cAAc,KAAK,SAAS,CACrD,EAAE,UAAU,CAAC,cAAc,CAAC;YAC7B,MAAM,MAAM,GACV,UAAU,GAAG,CAAC;gBACZ,CAAC,CAAC,OAAO,GAAG,mCAAmC,UAAU,cAAc;oBACrE,CAAC,KAAK,KAAK,SAAS,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,YAAY,KAAK,GAAG,CAAC;oBACjD,sHAAsH;gBACxH,CAAC,CAAC,OAAO,GAAG,kCAAkC,MAAM,CAAC,IAAI,CAAC,MAAM,mGAAmG,CAAC;YACxK,QAAQ,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,YAAY,EAAE,MAAM,EAAE,CAAC,CAAC;QAChD,CAAC;IACH,CAAC;IAED,0EAA0E;IAC1E,EAAE;IACF,0EAA0E;IAC1E,wEAAwE;IACxE,sEAAsE;IACtE,MAAM,cAAc,GAAG,IAAI,GAAG,CAC5B,OAAO,KAAK,SAAS,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,OAAO,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,IAAI,CAAC,cAAc,CAAC,CAChF,CAAC;IACF,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,WAAW,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC;QAC5D,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,IAAI,EAAE,CAAC;YAC9B,IAAI,CAAC,cAAc,CAAC,GAAG,CAAC,GAAG,CAAC,IAAI,CAAC,cAAc,CAAC,EAAE,CAAC;gBACjD,QAAQ,CAAC,IAAI,CAAC;oBACZ,IAAI,EAAE,kBAAkB;oBACxB,MAAM,EACJ,OAAO,GAAG,UAAU,GAAG,CAAC,KAAK,6BAA6B,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,cAAc,CAAC,GAAG;wBAC5F,sCAAsC,CAAC,GAAG,cAAc,CAAC,CAAC,GAAG,CAAC,MAAM,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,GAAG;iBACtF,CAAC,CAAC;YACL,CAAC;QACH,CAAC;IACH,CAAC;IAED,2EAA2E;IAC3E,mEAAmE;IACnE,mEAAmE;IACnE,MAAM,QAAQ,GAAG,QAAQ,CAAC;IAC1B,IAAI,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACxB,MAAM,KAAK,GAAG,QAAQ,CAAC,IAAI,CAAC,CAAC,OAAO,EAAE,EAAE,CAAC,OAAO,CAAC,IAAI,KAAK,YAAY,CAAC,CAAC;QACxE,MAAM,OAAO,GAAG,QAAQ,CAAC,IAAI,CAC3B,CAAC,OAAO,EAAE,EAAE,CACV,OAAO,CAAC,IAAI,KAAK,eAAe;YAChC,OAAO,CAAC,IAAI,KAAK,cAAc;YAC/B,OAAO,CAAC,IAAI,KAAK,aAAa,CACjC,CAAC;QACF,OAAO,MAAM,CACX,KAAK,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,gBAAgB,EACxD,QAAQ,EACR,aAAa,CACd,CAAC;IACJ,CAAC;IAED,0EAA0E;IAC1E,MAAM,CAAC,eAAe,EAAE,kBAAkB,CAAC,GAAG,YAAY,CAAC,CAAC,CAAE,CAAC;IAC/D,MAAM,MAAM,GAAG,UAAU,CACvB,OAAQ,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,YAAY,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,EACnD,kBAAkB,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,YAAY,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,CAC9D,CAAC;IAEF,IAAI,MAAM,CAAC,MAAM,KAAK,IAAI,IAAI,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,MAAM,CAAC,GAAG,kBAAkB,EAAE,CAAC;QAC3E,QAAQ,CAAC,IAAI,CAAC;YACZ,IAAI,EAAE,UAAU;YAChB,MAAM,EACJ,aAAa,MAAM,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC,CAAC,oBAAoB,kBAAkB,gCAAgC;gBAC3G,QAAQ,eAAe,yBAAyB,CAAC,aAAa,CAAC,eAAe,CAAC,CAAC,WAAW,GAAG,GAAG,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,iBAAiB;SACjI,CAAC,CAAC;IACL,CAAC;IAED,4EAA4E;IAC5E,IAAI,QAAQ,CAAC,MAAM,KAAK,CAAC,IAAI,OAAO,CAAC,mBAAmB,KAAK,IAAI,EAAE,CAAC;QAClE,QAAQ,CAAC,IAAI,CAAC;YACZ,IAAI,EAAE,qBAAqB;YAC3B,MAAM,EACJ,uHAAuH;gBACvH,0GAA0G;gBAC1G,iEAAiE;SACpE,CAAC,CAAC;IACL,CAAC;IAED,IAAI,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACxB,wEAAwE;QACxE,qEAAqE;QACrE,uEAAuE;QACvE,iEAAiE;QACjE,MAAM,gBAAgB,GACpB,OAAO,CAAC,mBAAmB,KAAK,IAAI;YACpC,QAAQ,CAAC,CAAC,CAAE,CAAC,IAAI,KAAK,qBAAqB,CAAC;QAC9C,OAAO;YACL,OAAO,EAAE,WAAW;YACpB,QAAQ;YACR,aAAa;YACb,MAAM;YACN,OAAO,EAAE,GAAG,gBAAgB,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,WAAW,KAAK,QAAQ,CAAC,CAAC,CAAE,CAAC,MAAM,KAAK,QAAQ,CAAC,MAAM,kBAAkB;SACxH,CAAC;IACJ,CAAC;IAED,MAAM,SAAS,GAAG,MAAM,CAAC,UAAU,GAAG,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC;IAC3D,MAAM,SAAS,GACb,MAAM,CAAC,MAAM,KAAK,IAAI;QACpB,CAAC,CAAC,kCAAkC;QACpC,CAAC,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC;IACnD,OAAO;QACL,OAAO,EAAE,MAAM,CAAC,UAAU,GAAG,CAAC,CAAC,CAAC,CAAC,QAAQ,CAAC,CAAC,CAAC,YAAY;QACxD,QAAQ;QACR,aAAa;QACb,MAAM;QACN,OAAO,EACL,GAAG,MAAM,CAAC,UAAU,GAAG,CAAC,CAAC,CAAC,CAAC,QAAQ,CAAC,CAAC,CAAC,YAAY,SAAS,eAAe,SAAS;YACnF,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,UAAU,CAAC,IAAI,SAAS,iCAAiC,SAAS,GAAG;KAC3F,CAAC;AACJ,CAAC;AAED,oFAAoF;AACpF,SAAS,MAAM,CACb,OAAgB,EAChB,QAAgC,EAChC,aAAoD;IAEpD,OAAO;QACL,OAAO;QACP,QAAQ;QACR,aAAa;QACb,MAAM,EAAE,IAAI;QACZ,OAAO,EAAE,GAAG,QAAQ,CAAC,OAAO,CAAC,KAAK,QAAQ,CAAC,CAAC,CAAE,CAAC,MAAM,KAAK,QAAQ,CAAC,MAAM,kBAAkB;KAC5F,CAAC;AACJ,CAAC;AAED,MAAM,QAAQ,GAAsC;IAClD,MAAM,EAAE,QAAQ;IAChB,UAAU,EAAE,YAAY;IACxB,WAAW,EAAE,WAAW;IACxB,KAAK,EAAE,OAAO;IACd,gBAAgB,EAAE,gBAAgB;IAClC,OAAO,EAAE,SAAS;CACnB,CAAC"}
@@ -0,0 +1,35 @@
1
+ /**
2
+ * `skillstate ab` — drive an A/B experiment and print a gated verdict.
3
+ *
4
+ * @non-paper — measurement infrastructure, not the paper's evaluation.
5
+ *
6
+ * Importing this module is side-effect free: it runs only when this file is
7
+ * the process entry (`node dist/ab-cli.js`), never on library import.
8
+ *
9
+ * ── The one rule this CLI enforces ───────────────────────────────────────
10
+ *
11
+ * It prints a percentage only when every gate passed. On any refusal it
12
+ * prints the gates instead. This is the whole point of the tool: the previous
13
+ * experiment's number was not wrong by arithmetic, it was wrong because
14
+ * nothing asked whether the instrumented arm had done anything.
15
+ */
16
+ import type { ExperimentOptions } from './ab/verdict.js';
17
+ /** Options for {@link main}. */
18
+ export interface AbOptions extends ExperimentOptions {
19
+ /** Run files to read, in any order. */
20
+ readonly files: readonly string[];
21
+ /** Mark the task as defined over the historical trajectory. */
22
+ readonly taskNeedsTranscript?: boolean;
23
+ /** Overrides the value inferred from the run files. */
24
+ readonly minTrials?: number;
25
+ }
26
+ /**
27
+ * Read run files, run the gates, and print the report.
28
+ *
29
+ * Returns a process exit code. A refusal exits non-zero so a CI job wired to
30
+ * this cannot record an invalid experiment as a passing one.
31
+ */
32
+ export declare function main(options: AbOptions): number;
33
+ /** Parse argv and run. Exported so tests can drive it without a subprocess. */
34
+ export declare function mainFromArgv(argv: readonly string[]): number;
35
+ //# sourceMappingURL=ab-cli.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"ab-cli.d.ts","sourceRoot":"","sources":["../src/ab-cli.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AAWH,OAAO,KAAK,EAAE,iBAAiB,EAAE,MAAM,iBAAiB,CAAC;AA0EzD,gCAAgC;AAChC,MAAM,WAAW,SAAU,SAAQ,iBAAiB;IAClD,uCAAuC;IACvC,QAAQ,CAAC,KAAK,EAAE,SAAS,MAAM,EAAE,CAAC;IAClC,+DAA+D;IAC/D,QAAQ,CAAC,mBAAmB,CAAC,EAAE,OAAO,CAAC;IACvC,uDAAuD;IACvD,QAAQ,CAAC,SAAS,CAAC,EAAE,MAAM,CAAC;CAC7B;AAED;;;;;GAKG;AACH,wBAAgB,IAAI,CAAC,OAAO,EAAE,SAAS,GAAG,MAAM,CAkC/C;AAED,+EAA+E;AAC/E,wBAAgB,YAAY,CAAC,IAAI,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,CAkD5D"}
package/dist/ab-cli.js ADDED
@@ -0,0 +1,161 @@
1
+ /**
2
+ * `skillstate ab` — drive an A/B experiment and print a gated verdict.
3
+ *
4
+ * @non-paper — measurement infrastructure, not the paper's evaluation.
5
+ *
6
+ * Importing this module is side-effect free: it runs only when this file is
7
+ * the process entry (`node dist/ab-cli.js`), never on library import.
8
+ *
9
+ * ── The one rule this CLI enforces ───────────────────────────────────────
10
+ *
11
+ * It prints a percentage only when every gate passed. On any refusal it
12
+ * prints the gates instead. This is the whole point of the tool: the previous
13
+ * experiment's number was not wrong by arithmetic, it was wrong because
14
+ * nothing asked whether the instrumented arm had done anything.
15
+ */
16
+ import * as fs from 'node:fs';
17
+ import * as path from 'node:path';
18
+ import { pathToFileURL } from 'node:url';
19
+ import { assessEngagement, isWitnessed, withRejections } from './ab/engagement.js';
20
+ import { buildArmRecord } from './ab/record.js';
21
+ import { formatArmTable, formatVerdict } from './ab/report.js';
22
+ import { runExperiment } from './ab/verdict.js';
23
+ function parseRunFiles(files) {
24
+ const parsed = [];
25
+ for (const file of files) {
26
+ const raw = JSON.parse(fs.readFileSync(file, 'utf-8'));
27
+ for (const entry of Array.isArray(raw) ? raw : [raw]) {
28
+ parsed.push(entry);
29
+ }
30
+ }
31
+ return parsed;
32
+ }
33
+ function toRunRecord(trial) {
34
+ const report = assessEngagement(trial.stateSamples);
35
+ const witnessed = isWitnessed(trial.stateSamples);
36
+ // An unwitnessed trial is NOT recorded as inert. Nobody watched the file,
37
+ // and calling that "the integration did nothing" would be an accusation
38
+ // the harness cannot support. It is recorded as unwitnessed, which the
39
+ // sample-size and engagement gates then treat as a refusal.
40
+ const engagement = witnessed
41
+ ? withRejections(report.evidence, trial.rejections ?? 0, trial.firstRejection)
42
+ : {
43
+ ...report.evidence,
44
+ engaged: false,
45
+ writes: 0,
46
+ rejections: 0,
47
+ };
48
+ return {
49
+ arm: trial.arm,
50
+ trial: trial.trial,
51
+ task: trial.task,
52
+ model: trial.model,
53
+ hostVersion: trial.hostVersion,
54
+ sessionID: trial.sessionID,
55
+ usage: trial.usage,
56
+ engagement,
57
+ work: {
58
+ artifactDigest: trial.artifactDigest,
59
+ turns: trial.turns,
60
+ toolCalls: trial.toolCalls,
61
+ },
62
+ durationMs: trial.durationMs,
63
+ completed: trial.completed,
64
+ ...(trial.correct === undefined ? {} : { outcome: { correct: trial.correct } }),
65
+ ...(trial.error === undefined ? {} : { error: trial.error }),
66
+ };
67
+ }
68
+ /**
69
+ * Read run files, run the gates, and print the report.
70
+ *
71
+ * Returns a process exit code. A refusal exits non-zero so a CI job wired to
72
+ * this cannot record an invalid experiment as a passing one.
73
+ */
74
+ export function main(options) {
75
+ const trials = parseRunFiles(options.files);
76
+ const byArm = new Map();
77
+ for (const trial of trials) {
78
+ const runs = byArm.get(trial.arm) ?? [];
79
+ runs.push(toRunRecord(trial));
80
+ byArm.set(trial.arm, runs);
81
+ }
82
+ const arms = new Map();
83
+ const problems = [];
84
+ for (const [armId, runs] of byArm) {
85
+ const built = buildArmRecord(armId, runs);
86
+ if (!built.ok) {
87
+ problems.push(built.reason);
88
+ continue;
89
+ }
90
+ arms.set(armId, built.record);
91
+ }
92
+ const result = runExperiment(arms, {
93
+ ...(options.taskNeedsTranscript === undefined
94
+ ? {}
95
+ : { taskNeedsTranscript: options.taskNeedsTranscript }),
96
+ ...(options.minTrials === undefined ? {} : { minTrials: options.minTrials }),
97
+ });
98
+ console.log(formatVerdict(result, formatArmTable(arms)));
99
+ for (const problem of problems) {
100
+ console.log(` - [input] ${problem}`);
101
+ }
102
+ const measured = result.verdict === 'saving' || result.verdict === 'regression';
103
+ return measured && problems.length === 0 ? 0 : 1;
104
+ }
105
+ /** Parse argv and run. Exported so tests can drive it without a subprocess. */
106
+ export function mainFromArgv(argv) {
107
+ const files = [];
108
+ let taskNeedsTranscript = false;
109
+ let minTrials;
110
+ for (let i = 0; i < argv.length; i += 1) {
111
+ const arg = argv[i];
112
+ if (arg === '--transcript-task') {
113
+ taskNeedsTranscript = true;
114
+ }
115
+ else if (arg === '--min-trials') {
116
+ const value = Number(argv[++i]);
117
+ if (!Number.isInteger(value) || value < 1) {
118
+ console.log('--min-trials needs a positive integer');
119
+ return 2;
120
+ }
121
+ minTrials = value;
122
+ }
123
+ else if (arg === '--help' || arg === '-h') {
124
+ console.log('usage: skillstate ab [--transcript-task] [--min-trials N] <run.json>...\n\n' +
125
+ 'Each run file is a JSON object (or array of them) describing one trial:\n' +
126
+ ' arm, trial, task, model, hostVersion, sessionID, usage{input,cacheRead,\n' +
127
+ ' cacheWrite,output}, turns, toolCalls, artifactDigest, durationMs,\n' +
128
+ ' completed, and stateSamples[{trial,step,content,fromSink?}].\n\n' +
129
+ 'Exits 0 only when every gate passed and a saving or regression was found.');
130
+ return 0;
131
+ }
132
+ else if (arg.startsWith('-')) {
133
+ console.log(`unknown flag: ${arg}`);
134
+ return 2;
135
+ }
136
+ else {
137
+ files.push(arg);
138
+ }
139
+ }
140
+ if (files.length === 0) {
141
+ console.log('no run files given — nothing to compare');
142
+ return 2;
143
+ }
144
+ for (const file of files) {
145
+ if (!fs.existsSync(path.resolve(file))) {
146
+ console.log(`no such run file: ${file}`);
147
+ return 2;
148
+ }
149
+ }
150
+ return main({
151
+ files,
152
+ ...(taskNeedsTranscript ? { taskNeedsTranscript: true } : {}),
153
+ ...(minTrials === undefined ? {} : { minTrials }),
154
+ });
155
+ }
156
+ const isEntry = process.argv[1] !== undefined &&
157
+ pathToFileURL(process.argv[1]).href === import.meta.url;
158
+ if (isEntry) {
159
+ process.exitCode = mainFromArgv(process.argv.slice(2));
160
+ }
161
+ //# sourceMappingURL=ab-cli.js.map