@skillstate/bench 3.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -2
- package/dist/ab/engagement.d.ts +70 -0
- package/dist/ab/engagement.d.ts.map +1 -0
- package/dist/ab/engagement.js +102 -0
- package/dist/ab/engagement.js.map +1 -0
- package/dist/ab/index.d.ts +28 -0
- package/dist/ab/index.d.ts.map +1 -0
- package/dist/ab/index.js +20 -0
- package/dist/ab/index.js.map +1 -0
- package/dist/ab/opencode-usage.d.ts +52 -0
- package/dist/ab/opencode-usage.d.ts.map +1 -0
- package/dist/ab/opencode-usage.js +85 -0
- package/dist/ab/opencode-usage.js.map +1 -0
- package/dist/ab/record.d.ts +173 -0
- package/dist/ab/record.d.ts.map +1 -0
- package/dist/ab/record.js +98 -0
- package/dist/ab/record.js.map +1 -0
- package/dist/ab/replay.d.ts +193 -0
- package/dist/ab/replay.d.ts.map +1 -0
- package/dist/ab/replay.js +149 -0
- package/dist/ab/replay.js.map +1 -0
- package/dist/ab/report.d.ts +26 -0
- package/dist/ab/report.d.ts.map +1 -0
- package/dist/ab/report.js +98 -0
- package/dist/ab/report.js.map +1 -0
- package/dist/ab/stats-core.d.ts +84 -0
- package/dist/ab/stats-core.d.ts.map +1 -0
- package/dist/ab/stats-core.js +93 -0
- package/dist/ab/stats-core.js.map +1 -0
- package/dist/ab/survey.d.ts +131 -0
- package/dist/ab/survey.d.ts.map +1 -0
- package/dist/ab/survey.js +143 -0
- package/dist/ab/survey.js.map +1 -0
- package/dist/ab/usage.d.ts +99 -0
- package/dist/ab/usage.d.ts.map +1 -0
- package/dist/ab/usage.js +98 -0
- package/dist/ab/usage.js.map +1 -0
- package/dist/ab/verdict.d.ts +88 -0
- package/dist/ab/verdict.d.ts.map +1 -0
- package/dist/ab/verdict.js +251 -0
- package/dist/ab/verdict.js.map +1 -0
- package/dist/ab-cli.d.ts +35 -0
- package/dist/ab-cli.d.ts.map +1 -0
- package/dist/ab-cli.js +161 -0
- package/dist/ab-cli.js.map +1 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/package.json +5 -1
package/dist/ab/usage.js
ADDED
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Token accounting read from the HOST's own store.
|
|
3
|
+
*
|
|
4
|
+
* ── Why not parse the host's stdout ──────────────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* Three reasons, in order of how much they matter:
|
|
7
|
+
*
|
|
8
|
+
* 1. **The model does not report its own cost reliably.** A harness that
|
|
9
|
+
* takes the number the subject of the experiment wrote down is trusting the
|
|
10
|
+
* measurement to the thing being measured. The previous A/B took its
|
|
11
|
+
* figures from the run's own reporting, and one of its two "arms" was
|
|
12
|
+
* itself two different readings of a single session.
|
|
13
|
+
* 2. **OpenCode already records per-message token usage** in its local store
|
|
14
|
+
* (`message.data.tokens` with `input`, `output`, `reasoning` and
|
|
15
|
+
* `cache.{read,write}`). That is the host's own accounting, not a
|
|
16
|
+
* reconstruction, and it is keyed by session so a run can be re-read
|
|
17
|
+
* after the fact.
|
|
18
|
+
* 3. **It is available without a live model.** Reading the store works when
|
|
19
|
+
* the provider quota is exhausted, which is exactly when one most wants to
|
|
20
|
+
* re-analyse a previous run.
|
|
21
|
+
*
|
|
22
|
+
* Node's `node:sqlite` is used rather than a dependency: the store is a
|
|
23
|
+
* local file, the read is a single prepared statement, and adding a native
|
|
24
|
+
* dependency to a zero-deps package for this would be a poor trade. The
|
|
25
|
+
* reader is injected, so the query and its failure modes are testable
|
|
26
|
+
* without an OpenCode installation.
|
|
27
|
+
*
|
|
28
|
+
* @non-paper — measurement infrastructure, not the paper's evaluation.
|
|
29
|
+
*/
|
|
30
|
+
/** Zero usage, for a session whose messages carry no accounting. */
|
|
31
|
+
export const NO_USAGE = {
|
|
32
|
+
input: 0,
|
|
33
|
+
cacheRead: 0,
|
|
34
|
+
cacheWrite: 0,
|
|
35
|
+
output: 0,
|
|
36
|
+
};
|
|
37
|
+
/** Sum one row's token fields, tolerating a partial or absent record. */
|
|
38
|
+
function rowUsage(tokens) {
|
|
39
|
+
return {
|
|
40
|
+
input: tokens?.input ?? 0,
|
|
41
|
+
cacheRead: tokens?.cacheRead ?? 0,
|
|
42
|
+
cacheWrite: tokens?.cacheWrite ?? 0,
|
|
43
|
+
output: tokens?.output ?? 0,
|
|
44
|
+
};
|
|
45
|
+
}
|
|
46
|
+
/**
|
|
47
|
+
* Total token spend for a session.
|
|
48
|
+
*
|
|
49
|
+
* Only `assistant` rows count. A user message's `input` figure is the host
|
|
50
|
+
* describing the prompt it echoed, and adding it would double-count the
|
|
51
|
+
* conversation.
|
|
52
|
+
*
|
|
53
|
+
* `cacheRead` is summed, not treated as free: prompt caching changes the
|
|
54
|
+
* price of a token, not the fact that the model was shown it, and the
|
|
55
|
+
* paper's claim is about what the model is exposed to.
|
|
56
|
+
*/
|
|
57
|
+
export function sessionUsage(rows) {
|
|
58
|
+
return rows
|
|
59
|
+
.filter((row) => row.role === 'assistant')
|
|
60
|
+
.map((row) => rowUsage(row.tokens))
|
|
61
|
+
.reduce((sum, usage) => ({
|
|
62
|
+
input: sum.input + usage.input,
|
|
63
|
+
cacheRead: sum.cacheRead + usage.cacheRead,
|
|
64
|
+
cacheWrite: sum.cacheWrite + usage.cacheWrite,
|
|
65
|
+
output: sum.output + usage.output,
|
|
66
|
+
}), NO_USAGE);
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
69
|
+
* Resolve a session's token spend, turning every failure into a value.
|
|
70
|
+
*
|
|
71
|
+
* A run whose accounting cannot be read is a run with unknown cost, which is
|
|
72
|
+
* not the same as a run that cost nothing. The distinction is the whole
|
|
73
|
+
* reason this returns a union instead of a number: `store_unavailable` must
|
|
74
|
+
* never be silently read as `NO_USAGE`.
|
|
75
|
+
*/
|
|
76
|
+
export async function resolveSessionUsage(reader, sessionID) {
|
|
77
|
+
let rows;
|
|
78
|
+
try {
|
|
79
|
+
rows = await reader.messagesFor(sessionID);
|
|
80
|
+
}
|
|
81
|
+
catch (error) {
|
|
82
|
+
return {
|
|
83
|
+
ok: false,
|
|
84
|
+
reason: 'store_unavailable',
|
|
85
|
+
detail: error instanceof Error ? error.message : String(error),
|
|
86
|
+
};
|
|
87
|
+
}
|
|
88
|
+
if (rows.length === 0) {
|
|
89
|
+
return {
|
|
90
|
+
ok: false,
|
|
91
|
+
reason: 'session_unknown',
|
|
92
|
+
detail: `no messages recorded for session ${sessionID}`,
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
const assistant = rows.filter((row) => row.role === 'assistant').length;
|
|
96
|
+
return { ok: true, usage: sessionUsage(rows), assistantMessages: assistant };
|
|
97
|
+
}
|
|
98
|
+
//# sourceMappingURL=usage.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"usage.js","sourceRoot":"","sources":["../../src/ab/usage.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA4BG;AAmCH,oEAAoE;AACpE,MAAM,CAAC,MAAM,QAAQ,GAAe;IAClC,KAAK,EAAE,CAAC;IACR,SAAS,EAAE,CAAC;IACZ,UAAU,EAAE,CAAC;IACb,MAAM,EAAE,CAAC;CACV,CAAC;AAEF,yEAAyE;AACzE,SAAS,QAAQ,CAAC,MAAkC;IAClD,OAAO;QACL,KAAK,EAAE,MAAM,EAAE,KAAK,IAAI,CAAC;QACzB,SAAS,EAAE,MAAM,EAAE,SAAS,IAAI,CAAC;QACjC,UAAU,EAAE,MAAM,EAAE,UAAU,IAAI,CAAC;QACnC,MAAM,EAAE,MAAM,EAAE,MAAM,IAAI,CAAC;KAC5B,CAAC;AACJ,CAAC;AAED;;;;;;;;;;GAUG;AACH,MAAM,UAAU,YAAY,CAAC,IAA+B;IAC1D,OAAO,IAAI;SACR,MAAM,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,IAAI,KAAK,WAAW,CAAC;SACzC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,QAAQ,CAAC,GAAG,CAAC,MAAM,CAAC,CAAC;SAClC,MAAM,CACL,CAAC,GAAG,EAAE,KAAK,EAAE,EAAE,CAAC,CAAC;QACf,KAAK,EAAE,GAAG,CAAC,KAAK,GAAG,KAAK,CAAC,KAAK;QAC9B,SAAS,EAAE,GAAG,CAAC,SAAS,GAAG,KAAK,CAAC,SAAS;QAC1C,UAAU,EAAE,GAAG,CAAC,UAAU,GAAG,KAAK,CAAC,UAAU;QAC7C,MAAM,EAAE,GAAG,CAAC,MAAM,GAAG,KAAK,CAAC,MAAM;KAClC,CAAC,EACF,QAAQ,CACT,CAAC;AACN,CAAC;AAcD;;;;;;;GAOG;AACH,MAAM,CAAC,KAAK,UAAU,mBAAmB,CACvC,MAAmB,EACnB,SAAiB;IAEjB,IAAI,IAA+B,CAAC;IACpC,IAAI,CAAC;QACH,IAAI,GAAG,MAAM,MAAM,CAAC,WAAW,CAAC,SAAS,CAAC,CAAC;IAC7C,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,OAAO;YACL,EAAE,EAAE,KAAK;YACT,MAAM,EAAE,mBAAmB;YAC3B,MAAM,EAAE,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC;SAC/D,CAAC;IACJ,CAAC;IACD,IAAI,IAAI,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACtB,OAAO;YACL,EAAE,EAAE,KAAK;YACT,MAAM,EAAE,iBAAiB;YACzB,MAAM,EAAE,oCAAoC,SAAS,EAAE;SACxD,CAAC;IACJ,CAAC;IACD,MAAM,SAAS,GAAG,IAAI,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,IAAI,KAAK,WAAW,CAAC,CAAC,MAAM,CAAC;IACxE,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,KAAK,EAAE,YAAY,CAAC,IAAI,CAAC,EAAE,iBAAiB,EAAE,SAAS,EAAE,CAAC;AAC/E,CAAC"}
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The gates: what has to be true before a token comparison may be reported.
|
|
3
|
+
*
|
|
4
|
+
* ── What went wrong the first time ───────────────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* The previous A/B reported a 39% saving from a run in which the
|
|
7
|
+
* instrumented arm never once wrote the state file. Nothing flagged it,
|
|
8
|
+
* because the pipeline carried a number and a number has no opinion about
|
|
9
|
+
* whether the thing it measured was switched on.
|
|
10
|
+
*
|
|
11
|
+
* So the harness's primary output is a {@link Verdict}, and a bare percentage
|
|
12
|
+
* is only reachable by passing every gate. The gates are deliberately
|
|
13
|
+
* pessimistic and deliberately ordered: the most invalidating finding is
|
|
14
|
+
* computed first, so a later gate cannot refine a comparison that was never
|
|
15
|
+
* a comparison.
|
|
16
|
+
*
|
|
17
|
+
* @non-paper — measurement infrastructure for OUR host integration, not the
|
|
18
|
+
* paper's own evaluation. The paper's Limitations are cited here as the
|
|
19
|
+
* source of one gate, not as a claim being verified.
|
|
20
|
+
*/
|
|
21
|
+
import type { ArmId, ArmRecord } from './record.js';
|
|
22
|
+
import type { Distribution, EffectSize } from './stats-core.js';
|
|
23
|
+
/** Minimum trials per arm before a comparison may claim anything. */
|
|
24
|
+
export declare const MIN_TRIALS = 2;
|
|
25
|
+
/**
|
|
26
|
+
* What the harness concluded.
|
|
27
|
+
*
|
|
28
|
+
* The last four are refusals, and refusals are the point: a harness that can
|
|
29
|
+
* only produce numbers is a harness that will produce the wrong number.
|
|
30
|
+
*/
|
|
31
|
+
export type Verdict =
|
|
32
|
+
/** Engaged, comparable, and the instrumented arm spent significantly fewer prompt tokens. */
|
|
33
|
+
'saving'
|
|
34
|
+
/** Engaged, comparable, and the instrumented arm spent significantly more. */
|
|
35
|
+
| 'regression'
|
|
36
|
+
/** A valid run whose effect sits inside the noise. */
|
|
37
|
+
| 'no-effect'
|
|
38
|
+
/** The instrumented arm never wrote state. No number is reported. */
|
|
39
|
+
| 'inert'
|
|
40
|
+
/** The arms did not do comparable work, so their tokens are not comparable. */
|
|
41
|
+
| 'not-comparable'
|
|
42
|
+
/** A failed run, a mismatched task/model/host, or too few trials. */
|
|
43
|
+
| 'invalid';
|
|
44
|
+
/** A single gate that refused to let a number be reported. */
|
|
45
|
+
export interface GateFailure {
|
|
46
|
+
readonly gate: 'comparability' | 'completeness' | 'sample-size' | 'engagement' | 'task-equivalence' | 'outcome' | 'variance' | 'paper-compatibility';
|
|
47
|
+
/** Why the gate fired, phrased so a reader can act on it. */
|
|
48
|
+
readonly detail: string;
|
|
49
|
+
}
|
|
50
|
+
export interface VerdictResult {
|
|
51
|
+
readonly verdict: Verdict;
|
|
52
|
+
/** Every gate that fired, in gate order. */
|
|
53
|
+
readonly failures: readonly GateFailure[];
|
|
54
|
+
/** Per-arm prompt-token distributions; empty arms carry `n: 0`. */
|
|
55
|
+
readonly distributions: Readonly<Record<ArmId, Distribution>>;
|
|
56
|
+
/** The effect size, only when both arms have usable samples. */
|
|
57
|
+
readonly effect: EffectSize | null;
|
|
58
|
+
/** One line for printing. Never a bare percentage on a refusal. */
|
|
59
|
+
readonly summary: string;
|
|
60
|
+
}
|
|
61
|
+
/** Options for {@link runExperiment}. */
|
|
62
|
+
export interface ExperimentOptions {
|
|
63
|
+
/**
|
|
64
|
+
* Set when the task's objective is defined over the historical trajectory
|
|
65
|
+
* (an audit, a review, a "what did I just do" task).
|
|
66
|
+
*
|
|
67
|
+
* Limitations, case (3): the paper's own assumption fails exactly there. The
|
|
68
|
+
* section is not numbered — `state.md` records it as `§ Limitations` with no
|
|
69
|
+
* number, so a number here would be a citation nobody can check. An
|
|
70
|
+
* unverifiable citation is worse than none: it looks verified.
|
|
71
|
+
* A flat result on such a task is the predicted outcome, so reporting it
|
|
72
|
+
* as a refutation would be a category error in the harness's own favour.
|
|
73
|
+
* The gate downgrades `no-effect` to a failure that says so.
|
|
74
|
+
*/
|
|
75
|
+
readonly taskNeedsTranscript?: boolean;
|
|
76
|
+
/** Trials required per arm. Defaults to {@link MIN_TRIALS}. */
|
|
77
|
+
readonly minTrials?: number;
|
|
78
|
+
}
|
|
79
|
+
/**
|
|
80
|
+
* Run every gate and return the strongest conclusion the data supports.
|
|
81
|
+
*
|
|
82
|
+
* Pure and deterministic: no clock, no filesystem, no randomness. Identical
|
|
83
|
+
* input yields an identical result, which is what lets the gates be tested
|
|
84
|
+
* against the real historical runs and against inputs built to trip each
|
|
85
|
+
* branch.
|
|
86
|
+
*/
|
|
87
|
+
export declare function runExperiment(arms: ReadonlyMap<ArmId, ArmRecord>, options?: ExperimentOptions): VerdictResult;
|
|
88
|
+
//# sourceMappingURL=verdict.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"verdict.d.ts","sourceRoot":"","sources":["../../src/ab/verdict.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAGH,OAAO,KAAK,EAAE,KAAK,EAAE,SAAS,EAAE,MAAM,aAAa,CAAC;AAEpD,OAAO,KAAK,EAAE,YAAY,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAEhE,qEAAqE;AACrE,eAAO,MAAM,UAAU,IAAI,CAAC;AAE5B;;;;;GAKG;AACH,MAAM,MAAM,OAAO;AACjB,6FAA6F;AAC3F,QAAQ;AACV,8EAA8E;GAC5E,YAAY;AACd,sDAAsD;GACpD,WAAW;AACb,qEAAqE;GACnE,OAAO;AACT,+EAA+E;GAC7E,gBAAgB;AAClB,qEAAqE;GACnE,SAAS,CAAC;AAEd,8DAA8D;AAC9D,MAAM,WAAW,WAAW;IAC1B,QAAQ,CAAC,IAAI,EACT,eAAe,GACf,cAAc,GACd,aAAa,GACb,YAAY,GACZ,kBAAkB,GAClB,SAAS,GACT,UAAU,GACV,qBAAqB,CAAC;IAC1B,6DAA6D;IAC7D,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;CACzB;AAED,MAAM,WAAW,aAAa;IAC5B,QAAQ,CAAC,OAAO,EAAE,OAAO,CAAC;IAC1B,4CAA4C;IAC5C,QAAQ,CAAC,QAAQ,EAAE,SAAS,WAAW,EAAE,CAAC;IAC1C,mEAAmE;IACnE,QAAQ,CAAC,aAAa,EAAE,QAAQ,CAAC,MAAM,CAAC,KAAK,EAAE,YAAY,CAAC,CAAC,CAAC;IAC9D,gEAAgE;IAChE,QAAQ,CAAC,MAAM,EAAE,UAAU,GAAG,IAAI,CAAC;IACnC,mEAAmE;IACnE,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;CAC1B;AAED,yCAAyC;AACzC,MAAM,WAAW,iBAAiB;IAChC;;;;;;;;;;;OAWG;IACH,QAAQ,CAAC,mBAAmB,CAAC,EAAE,OAAO,CAAC;IACvC,+DAA+D;IAC/D,QAAQ,CAAC,SAAS,CAAC,EAAE,MAAM,CAAC;CAC7B;AA0BD;;;;;;;GAOG;AACH,wBAAgB,aAAa,CAC3B,IAAI,EAAE,WAAW,CAAC,KAAK,EAAE,SAAS,CAAC,EACnC,OAAO,GAAE,iBAAsB,GAC9B,aAAa,CAiOf"}
|
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The gates: what has to be true before a token comparison may be reported.
|
|
3
|
+
*
|
|
4
|
+
* ── What went wrong the first time ───────────────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* The previous A/B reported a 39% saving from a run in which the
|
|
7
|
+
* instrumented arm never once wrote the state file. Nothing flagged it,
|
|
8
|
+
* because the pipeline carried a number and a number has no opinion about
|
|
9
|
+
* whether the thing it measured was switched on.
|
|
10
|
+
*
|
|
11
|
+
* So the harness's primary output is a {@link Verdict}, and a bare percentage
|
|
12
|
+
* is only reachable by passing every gate. The gates are deliberately
|
|
13
|
+
* pessimistic and deliberately ordered: the most invalidating finding is
|
|
14
|
+
* computed first, so a later gate cannot refine a comparison that was never
|
|
15
|
+
* a comparison.
|
|
16
|
+
*
|
|
17
|
+
* @non-paper — measurement infrastructure for OUR host integration, not the
|
|
18
|
+
* paper's own evaluation. The paper's Limitations are cited here as the
|
|
19
|
+
* source of one gate, not as a claim being verified.
|
|
20
|
+
*/
|
|
21
|
+
import { isInstrumented, promptTokens } from './record.js';
|
|
22
|
+
import { describe, effectSize, MIN_EFFECT_IN_MADS } from './stats-core.js';
|
|
23
|
+
/** Minimum trials per arm before a comparison may claim anything. */
|
|
24
|
+
export const MIN_TRIALS = 2;
|
|
25
|
+
const EMPTY_DISTRIBUTION = {
|
|
26
|
+
n: 0,
|
|
27
|
+
min: 0,
|
|
28
|
+
median: 0,
|
|
29
|
+
max: 0,
|
|
30
|
+
mad: 0,
|
|
31
|
+
relativeMad: 0,
|
|
32
|
+
sorted: [],
|
|
33
|
+
};
|
|
34
|
+
function distributionMap(arms) {
|
|
35
|
+
const out = {
|
|
36
|
+
plain: EMPTY_DISTRIBUTION,
|
|
37
|
+
notes: EMPTY_DISTRIBUTION,
|
|
38
|
+
paper: EMPTY_DISTRIBUTION,
|
|
39
|
+
};
|
|
40
|
+
for (const [arm, record] of arms) {
|
|
41
|
+
out[arm] = describe(record.runs.map((run) => promptTokens(run.usage)));
|
|
42
|
+
}
|
|
43
|
+
return out;
|
|
44
|
+
}
|
|
45
|
+
/**
|
|
46
|
+
* Run every gate and return the strongest conclusion the data supports.
|
|
47
|
+
*
|
|
48
|
+
* Pure and deterministic: no clock, no filesystem, no randomness. Identical
|
|
49
|
+
* input yields an identical result, which is what lets the gates be tested
|
|
50
|
+
* against the real historical runs and against inputs built to trip each
|
|
51
|
+
* branch.
|
|
52
|
+
*/
|
|
53
|
+
export function runExperiment(arms, options = {}) {
|
|
54
|
+
const minTrials = options.minTrials ?? MIN_TRIALS;
|
|
55
|
+
const failures = [];
|
|
56
|
+
const distributions = distributionMap(arms);
|
|
57
|
+
const control = arms.get('plain');
|
|
58
|
+
const instrumented = [...arms.entries()].filter(([arm]) => isInstrumented(arm));
|
|
59
|
+
// Every gate is evaluated before any early return, so a caller fixing a
|
|
60
|
+
// broken experiment is told everything that is wrong in one pass rather
|
|
61
|
+
// than discovering the next problem on the next run. Only gates that are
|
|
62
|
+
// undefined without a control arm (equivalence, effect) are skipped.
|
|
63
|
+
const hasBothArms = control !== undefined && instrumented.length > 0;
|
|
64
|
+
if (!hasBothArms) {
|
|
65
|
+
failures.push({
|
|
66
|
+
gate: 'comparability',
|
|
67
|
+
detail: 'need a `plain` control arm and at least one instrumented arm',
|
|
68
|
+
});
|
|
69
|
+
}
|
|
70
|
+
// ── Gate 1: comparability — same task, same model, same host ───────────
|
|
71
|
+
for (const [arm, record] of hasBothArms ? instrumented : []) {
|
|
72
|
+
const baseline = control;
|
|
73
|
+
if (record.task !== baseline.task ||
|
|
74
|
+
record.model !== baseline.model ||
|
|
75
|
+
record.hostVersion !== baseline.hostVersion) {
|
|
76
|
+
failures.push({
|
|
77
|
+
gate: 'comparability',
|
|
78
|
+
detail: `arm ${arm} ran ${JSON.stringify(record.task)} on ${record.model}@${record.hostVersion}` +
|
|
79
|
+
` but the control ran ${JSON.stringify(baseline.task)} on ${baseline.model}@${baseline.hostVersion}`,
|
|
80
|
+
});
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
// ── Gate 2: completeness — every run actually finished ────────────────
|
|
84
|
+
for (const [arm, record] of arms) {
|
|
85
|
+
for (const run of record.runs) {
|
|
86
|
+
if (!run.completed) {
|
|
87
|
+
failures.push({
|
|
88
|
+
gate: 'completeness',
|
|
89
|
+
detail: `arm ${arm} trial ${run.trial} did not complete: ${run.error ?? 'unknown error'}`,
|
|
90
|
+
});
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
// ── Gate 2b: the outcome — did the work get done? ───────────────────────
|
|
95
|
+
//
|
|
96
|
+
// The last gate, and the one that was missing. Every other gate asks whether
|
|
97
|
+
// the comparison is valid; none asks whether the result is worth anything.
|
|
98
|
+
// An arm that reads every file, writes state and never produces the answer
|
|
99
|
+
// passes engagement, comparability and variance, and would be reported as a
|
|
100
|
+
// saving. That is the failure this project has already paid for once: "paper
|
|
101
|
+
// 81,685 tokens vs control 203,801 = 60% cheaper, paper answered WRONG".
|
|
102
|
+
//
|
|
103
|
+
// Two refusals, and the second is the one that matters most. An experiment
|
|
104
|
+
// that measured cost and not the work has not earned the right to report the
|
|
105
|
+
// cost, whatever its token numbers say.
|
|
106
|
+
const measured = (record) => record.runs.some((run) => run.outcome !== undefined);
|
|
107
|
+
for (const [arm, record] of instrumented) {
|
|
108
|
+
if (!measured(record) || control === undefined || !measured(control)) {
|
|
109
|
+
failures.push({
|
|
110
|
+
gate: 'outcome',
|
|
111
|
+
detail: `arm ${arm} did not record whether the task was answered — a cost ` +
|
|
112
|
+
'saving with no outcome is not a result, so none is reported. Pass ' +
|
|
113
|
+
'`correct` in the trial files to measure it.',
|
|
114
|
+
});
|
|
115
|
+
continue;
|
|
116
|
+
}
|
|
117
|
+
const correct = (record) => record.runs.filter((run) => run.outcome?.correct === true).length;
|
|
118
|
+
// Strictly, not "at least as good". The n=7 experiment's honest reading
|
|
119
|
+
// was that the arms are comparably accurate and only the cost differs, and
|
|
120
|
+
// a gate that fired on a tie would have thrown that finding away.
|
|
121
|
+
if (correct(record) / record.runs.length < correct(control) / control.runs.length) {
|
|
122
|
+
failures.push({
|
|
123
|
+
gate: 'outcome',
|
|
124
|
+
detail: `arm ${arm} answered correctly ${correct(record)} of ${record.runs.length} trial(s) ` +
|
|
125
|
+
`while the control answered ${correct(control)} of ${control.runs.length} — ` +
|
|
126
|
+
'a cheaper run that did not finish the work is a regression, not a saving',
|
|
127
|
+
});
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
// ── Gate 3: sample size — one run cannot show variance ────────────────
|
|
131
|
+
for (const [arm, record] of arms) {
|
|
132
|
+
if (record.runs.length < minTrials) {
|
|
133
|
+
failures.push({
|
|
134
|
+
gate: 'sample-size',
|
|
135
|
+
detail: `arm ${arm} ran ${record.runs.length} trial(s); ${minTrials} are needed` +
|
|
136
|
+
' to separate an effect from run-to-run variance',
|
|
137
|
+
});
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
// ── Gate 4: engagement — was the integration switched on at all? ───────
|
|
141
|
+
//
|
|
142
|
+
// First in the list of reasons a run is unusable, because when it fires
|
|
143
|
+
// every token number in the experiment is a sample of model variance. The
|
|
144
|
+
// detail distinguishes "never engaged" from "engaged but every response
|
|
145
|
+
// was rejected", which are different bugs with different fixes.
|
|
146
|
+
for (const [arm, record] of instrumented) {
|
|
147
|
+
if (record.runs.every((run) => !run.engagement.engaged)) {
|
|
148
|
+
const rejections = record.runs.reduce((n, run) => n + run.engagement.rejections, 0);
|
|
149
|
+
const first = record.runs.find((run) => run.engagement.firstRejection !== undefined)?.engagement.firstRejection;
|
|
150
|
+
const detail = rejections > 0
|
|
151
|
+
? `arm ${arm} never wrote state and rejected ${rejections} response(s)` +
|
|
152
|
+
(first === undefined ? '' : ` (first: ${first})`) +
|
|
153
|
+
' — the model tried and the integration refused every patch, so its tokens measure the rejection, not the integration'
|
|
154
|
+
: `arm ${arm} never wrote the state file in ${record.runs.length} trial(s) — the integration was inert, so its token count is not a measurement of the integration`;
|
|
155
|
+
failures.push({ gate: 'engagement', detail });
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
// ── Gate 5: task equivalence — did both arms do the same work? ─────────
|
|
159
|
+
//
|
|
160
|
+
// Byte-identical artifacts are the strongest available evidence that both
|
|
161
|
+
// arms finished the same thing. A digest that appears in no control run
|
|
162
|
+
// means the arms are not comparable, whatever their token counts say.
|
|
163
|
+
const controlDigests = new Set(control === undefined ? [] : control.runs.map((run) => run.work.artifactDigest));
|
|
164
|
+
for (const [arm, record] of hasBothArms ? instrumented : []) {
|
|
165
|
+
for (const run of record.runs) {
|
|
166
|
+
if (!controlDigests.has(run.work.artifactDigest)) {
|
|
167
|
+
failures.push({
|
|
168
|
+
gate: 'task-equivalence',
|
|
169
|
+
detail: `arm ${arm} trial ${run.trial} produced artifact digest ${String(run.work.artifactDigest)},` +
|
|
170
|
+
` which is not among the control's (${[...controlDigests].map(String).join(', ')})`,
|
|
171
|
+
});
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
// Gates 1, 2, 3 and 5 all invalidate the comparison itself. Engagement (4)
|
|
176
|
+
// gets its own verdict, because "the integration did nothing" is a
|
|
177
|
+
// materially different finding from "the arms did different work".
|
|
178
|
+
const blocking = failures;
|
|
179
|
+
if (blocking.length > 0) {
|
|
180
|
+
const inert = blocking.some((failure) => failure.gate === 'engagement');
|
|
181
|
+
const invalid = blocking.some((failure) => failure.gate === 'comparability' ||
|
|
182
|
+
failure.gate === 'completeness' ||
|
|
183
|
+
failure.gate === 'sample-size');
|
|
184
|
+
return refuse(inert ? 'inert' : invalid ? 'invalid' : 'not-comparable', blocking, distributions);
|
|
185
|
+
}
|
|
186
|
+
// ── Gate 6: variance — is the effect bigger than the noise? ────────────
|
|
187
|
+
const [instrumentedArm, instrumentedRecord] = instrumented[0];
|
|
188
|
+
const effect = effectSize(control.runs.map((run) => promptTokens(run.usage)), instrumentedRecord.runs.map((run) => promptTokens(run.usage)));
|
|
189
|
+
if (effect.inMads !== null && Math.abs(effect.inMads) < MIN_EFFECT_IN_MADS) {
|
|
190
|
+
failures.push({
|
|
191
|
+
gate: 'variance',
|
|
192
|
+
detail: `effect is ${effect.inMads.toFixed(2)} MADs, under the ${MIN_EFFECT_IN_MADS} required to call it a signal;` +
|
|
193
|
+
` the ${instrumentedArm} arm's own spread is ±${(distributions[instrumentedArm].relativeMad * 100).toFixed(0)}% of its median`,
|
|
194
|
+
});
|
|
195
|
+
}
|
|
196
|
+
// ── Gate 7: paper compatibility — a null on the wrong task is not a null ─
|
|
197
|
+
if (failures.length === 0 && options.taskNeedsTranscript === true) {
|
|
198
|
+
failures.push({
|
|
199
|
+
gate: 'paper-compatibility',
|
|
200
|
+
detail: 'this task is defined over the historical trajectory (audit-style), which is the case the paper’s Limitations call out' +
|
|
201
|
+
' predicts will not benefit from a bounded prompt; a flat result here is consistent with the paper, not a' +
|
|
202
|
+
' refutation of it — measure a task that needs cross-turn memory',
|
|
203
|
+
});
|
|
204
|
+
}
|
|
205
|
+
if (failures.length > 0) {
|
|
206
|
+
// A flat result on a transcript-shaped task is not a measurement of the
|
|
207
|
+
// integration at all, so it gets its own headline: a reader who sees
|
|
208
|
+
// "NO-EFFECT" would reasonably conclude skillstate does not help, when
|
|
209
|
+
// the honest statement is that this task was never a test of it.
|
|
210
|
+
const onTranscriptTask = options.taskNeedsTranscript === true &&
|
|
211
|
+
failures[0].gate === 'paper-compatibility';
|
|
212
|
+
return {
|
|
213
|
+
verdict: 'no-effect',
|
|
214
|
+
failures,
|
|
215
|
+
distributions,
|
|
216
|
+
effect,
|
|
217
|
+
summary: `${onTranscriptTask ? 'NOT-A-TEST' : 'NO-EFFECT'}: ${failures[0].detail} [${failures.length} gate(s) failed]`,
|
|
218
|
+
};
|
|
219
|
+
}
|
|
220
|
+
const direction = effect.difference > 0 ? 'fewer' : 'more';
|
|
221
|
+
const magnitude = effect.inMads === null
|
|
222
|
+
? 'exact: both arms had zero spread'
|
|
223
|
+
: `${Math.abs(effect.inMads).toFixed(1)} MADs`;
|
|
224
|
+
return {
|
|
225
|
+
verdict: effect.difference > 0 ? 'saving' : 'regression',
|
|
226
|
+
failures,
|
|
227
|
+
distributions,
|
|
228
|
+
effect,
|
|
229
|
+
summary: `${effect.difference > 0 ? 'SAVING' : 'REGRESSION'}: arm ${instrumentedArm} spent ` +
|
|
230
|
+
`${Math.abs(effect.difference)} ${direction} prompt tokens at the median (${magnitude})`,
|
|
231
|
+
};
|
|
232
|
+
}
|
|
233
|
+
/** Build a refusal result. Every gate failure is reported, never just the first. */
|
|
234
|
+
function refuse(verdict, failures, distributions) {
|
|
235
|
+
return {
|
|
236
|
+
verdict,
|
|
237
|
+
failures,
|
|
238
|
+
distributions,
|
|
239
|
+
effect: null,
|
|
240
|
+
summary: `${HEADLINE[verdict]}: ${failures[0].detail} [${failures.length} gate(s) failed]`,
|
|
241
|
+
};
|
|
242
|
+
}
|
|
243
|
+
const HEADLINE = {
|
|
244
|
+
saving: 'SAVING',
|
|
245
|
+
regression: 'REGRESSION',
|
|
246
|
+
'no-effect': 'NO-EFFECT',
|
|
247
|
+
inert: 'INERT',
|
|
248
|
+
'not-comparable': 'NOT-COMPARABLE',
|
|
249
|
+
invalid: 'INVALID',
|
|
250
|
+
};
|
|
251
|
+
//# sourceMappingURL=verdict.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"verdict.js","sourceRoot":"","sources":["../../src/ab/verdict.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH,OAAO,EAAE,cAAc,EAAE,YAAY,EAAE,MAAM,aAAa,CAAC;AAE3D,OAAO,EAAE,QAAQ,EAAE,UAAU,EAAE,kBAAkB,EAAE,MAAM,iBAAiB,CAAC;AAG3E,qEAAqE;AACrE,MAAM,CAAC,MAAM,UAAU,GAAG,CAAC,CAAC;AAoE5B,MAAM,kBAAkB,GAAiB;IACvC,CAAC,EAAE,CAAC;IACJ,GAAG,EAAE,CAAC;IACN,MAAM,EAAE,CAAC;IACT,GAAG,EAAE,CAAC;IACN,GAAG,EAAE,CAAC;IACN,WAAW,EAAE,CAAC;IACd,MAAM,EAAE,EAAE;CACX,CAAC;AAEF,SAAS,eAAe,CACtB,IAAmC;IAEnC,MAAM,GAAG,GAAgC;QACvC,KAAK,EAAE,kBAAkB;QACzB,KAAK,EAAE,kBAAkB;QACzB,KAAK,EAAE,kBAAkB;KAC1B,CAAC;IACF,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,IAAI,EAAE,CAAC;QACjC,GAAG,CAAC,GAAG,CAAC,GAAG,QAAQ,CAAC,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,YAAY,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC;IACzE,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,aAAa,CAC3B,IAAmC,EACnC,OAAO,GAAsB,EAAE;IAE/B,MAAM,SAAS,GAAG,OAAO,CAAC,SAAS,IAAI,UAAU,CAAC;IAClD,MAAM,QAAQ,GAAkB,EAAE,CAAC;IACnC,MAAM,aAAa,GAAG,eAAe,CAAC,IAAI,CAAC,CAAC;IAE5C,MAAM,OAAO,GAAG,IAAI,CAAC,GAAG,CAAC,OAAO,CAAC,CAAC;IAClC,MAAM,YAAY,GAAG,CAAC,GAAG,IAAI,CAAC,OAAO,EAAE,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,GAAG,CAAC,EAAE,EAAE,CAAC,cAAc,CAAC,GAAG,CAAC,CAAC,CAAC;IAEhF,wEAAwE;IACxE,wEAAwE;IACxE,yEAAyE;IACzE,qEAAqE;IACrE,MAAM,WAAW,GAAG,OAAO,KAAK,SAAS,IAAI,YAAY,CAAC,MAAM,GAAG,CAAC,CAAC;IACrE,IAAI,CAAC,WAAW,EAAE,CAAC;QACjB,QAAQ,CAAC,IAAI,CAAC;YACZ,IAAI,EAAE,eAAe;YACrB,MAAM,EAAE,8DAA8D;SACvE,CAAC,CAAC;IACL,CAAC;IAED,0EAA0E;IAC1E,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,WAAW,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC;QAC5D,MAAM,QAAQ,GAAG,OAAQ,CAAC;QAC1B,IACE,MAAM,CAAC,IAAI,KAAK,QAAQ,CAAC,IAAI;YAC7B,MAAM,CAAC,KAAK,KAAK,QAAQ,CAAC,KAAK;YAC/B,MAAM,CAAC,WAAW,KAAK,QAAQ,CAAC,WAAW,EAC3C,CAAC;YACD,QAAQ,CAAC,IAAI,CAAC;gBACZ,IAAI,EAAE,eAAe;gBACrB,MAAM,EACJ,OAAO,GAAG,QAAQ,IAAI,CAAC,SAAS,CAAC,MAAM,CAAC,IAAI,CAAC,OAAO,MAAM,CAAC,KAAK,IAAI,MAAM,CAAC,WAAW,EAAE;oBACxF,wBAAwB,IAAI,CAAC,SAAS,CAAC,QAAQ,CAAC,IAAI,CAAC,OAAO,QAAQ,CAAC,KAAK,IAAI,QAAQ,CAAC,WAAW,EAAE;aACvG,CAAC,CAAC;QACL,CAAC;IACH,CAAC;IAED,yEAAyE;IACzE,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,IAAI,EAAE,CAAC;QACjC,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,IAAI,EAAE,CAAC;YAC9B,IAAI,CAAC,GAAG,CAAC,SAAS,EAAE,CAAC;gBACnB,QAAQ,CAAC,IAAI,CAAC;oBACZ,IAAI,EAAE,cAAc;oBACpB,MAAM,EAAE,OAAO,GAAG,UAAU,GAAG,CAAC,KAAK,sBAAsB,GAAG,CAAC,KAAK,IAAI,eAAe,EAAE;iBAC1F,CAAC,CAAC;YACL,CAAC;QACH,CAAC;IACH,CAAC;IAED,2EAA2E;IAC3E,EAAE;IACF,6EAA6E;IAC7E,2EAA2E;IAC3E,2EAA2E;IAC3E,4EAA4E;IAC5E,6EAA6E;IAC7E,yEAAyE;IACzE,EAAE;IACF,2EAA2E;IAC3E,6EAA6E;IAC7E,wCAAwC;IACxC,MAAM,QAAQ,GAAG,CAAC,MAAiB,EAAW,EAAE,CAC9C,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,OAAO,KAAK,SAAS,CAAC,CAAC;IACvD,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,YAAY,EAAE,CAAC;QACzC,IAAI,CAAC,QAAQ,CAAC,MAAM,CAAC,IAAI,OAAO,KAAK,SAAS,IAAI,CAAC,QAAQ,CAAC,OAAO,CAAC,EAAE,CAAC;YACrE,QAAQ,CAAC,IAAI,CAAC;gBACZ,IAAI,EAAE,SAAS;gBACf,MAAM,EACJ,OAAO,GAAG,yDAAyD;oBACnE,oEAAoE;oBACpE,6CAA6C;aAChD,CAAC,CAAC;YACH,SAAS;QACX,CAAC;QACD,MAAM,OAAO,GAAG,CAAC,MAAiB,EAAU,EAAE,CAC5C,MAAM,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,OAAO,EAAE,OAAO,KAAK,IAAI,CAAC,CAAC,MAAM,CAAC;QACpE,wEAAwE;QACxE,2EAA2E;QAC3E,kEAAkE;QAClE,IAAI,OAAO,CAAC,MAAM,CAAC,GAAG,MAAM,CAAC,IAAI,CAAC,MAAM,GAAG,OAAO,CAAC,OAAO,CAAC,GAAG,OAAO,CAAC,IAAI,CAAC,MAAM,EAAE,CAAC;YAClF,QAAQ,CAAC,IAAI,CAAC;gBACZ,IAAI,EAAE,SAAS;gBACf,MAAM,EACJ,OAAO,GAAG,uBAAuB,OAAO,CAAC,MAAM,CAAC,OAAO,MAAM,CAAC,IAAI,CAAC,MAAM,YAAY;oBACrF,8BAA8B,OAAO,CAAC,OAAO,CAAC,OAAO,OAAO,CAAC,IAAI,CAAC,MAAM,KAAK;oBAC7E,0EAA0E;aAC7E,CAAC,CAAC;QACL,CAAC;IACH,CAAC;IAED,yEAAyE;IACzE,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,IAAI,EAAE,CAAC;QACjC,IAAI,MAAM,CAAC,IAAI,CAAC,MAAM,GAAG,SAAS,EAAE,CAAC;YACnC,QAAQ,CAAC,IAAI,CAAC;gBACZ,IAAI,EAAE,aAAa;gBACnB,MAAM,EACJ,OAAO,GAAG,QAAQ,MAAM,CAAC,IAAI,CAAC,MAAM,cAAc,SAAS,aAAa;oBACxE,iDAAiD;aACpD,CAAC,CAAC;QACL,CAAC;IACH,CAAC;IAED,0EAA0E;IAC1E,EAAE;IACF,wEAAwE;IACxE,0EAA0E;IAC1E,wEAAwE;IACxE,gEAAgE;IAChE,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,YAAY,EAAE,CAAC;QACzC,IAAI,MAAM,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,UAAU,CAAC,OAAO,CAAC,EAAE,CAAC;YACxD,MAAM,UAAU,GAAG,MAAM,CAAC,IAAI,CAAC,MAAM,CACnC,CAAC,CAAC,EAAE,GAAG,EAAE,EAAE,CAAC,CAAC,GAAG,GAAG,CAAC,UAAU,CAAC,UAAU,EACzC,CAAC,CACF,CAAC;YACF,MAAM,KAAK,GAAG,MAAM,CAAC,IAAI,CAAC,IAAI,CAC5B,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,UAAU,CAAC,cAAc,KAAK,SAAS,CACrD,EAAE,UAAU,CAAC,cAAc,CAAC;YAC7B,MAAM,MAAM,GACV,UAAU,GAAG,CAAC;gBACZ,CAAC,CAAC,OAAO,GAAG,mCAAmC,UAAU,cAAc;oBACrE,CAAC,KAAK,KAAK,SAAS,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,YAAY,KAAK,GAAG,CAAC;oBACjD,sHAAsH;gBACxH,CAAC,CAAC,OAAO,GAAG,kCAAkC,MAAM,CAAC,IAAI,CAAC,MAAM,mGAAmG,CAAC;YACxK,QAAQ,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,YAAY,EAAE,MAAM,EAAE,CAAC,CAAC;QAChD,CAAC;IACH,CAAC;IAED,0EAA0E;IAC1E,EAAE;IACF,0EAA0E;IAC1E,wEAAwE;IACxE,sEAAsE;IACtE,MAAM,cAAc,GAAG,IAAI,GAAG,CAC5B,OAAO,KAAK,SAAS,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,OAAO,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,IAAI,CAAC,cAAc,CAAC,CAChF,CAAC;IACF,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,WAAW,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC;QAC5D,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,IAAI,EAAE,CAAC;YAC9B,IAAI,CAAC,cAAc,CAAC,GAAG,CAAC,GAAG,CAAC,IAAI,CAAC,cAAc,CAAC,EAAE,CAAC;gBACjD,QAAQ,CAAC,IAAI,CAAC;oBACZ,IAAI,EAAE,kBAAkB;oBACxB,MAAM,EACJ,OAAO,GAAG,UAAU,GAAG,CAAC,KAAK,6BAA6B,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,cAAc,CAAC,GAAG;wBAC5F,sCAAsC,CAAC,GAAG,cAAc,CAAC,CAAC,GAAG,CAAC,MAAM,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,GAAG;iBACtF,CAAC,CAAC;YACL,CAAC;QACH,CAAC;IACH,CAAC;IAED,2EAA2E;IAC3E,mEAAmE;IACnE,mEAAmE;IACnE,MAAM,QAAQ,GAAG,QAAQ,CAAC;IAC1B,IAAI,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACxB,MAAM,KAAK,GAAG,QAAQ,CAAC,IAAI,CAAC,CAAC,OAAO,EAAE,EAAE,CAAC,OAAO,CAAC,IAAI,KAAK,YAAY,CAAC,CAAC;QACxE,MAAM,OAAO,GAAG,QAAQ,CAAC,IAAI,CAC3B,CAAC,OAAO,EAAE,EAAE,CACV,OAAO,CAAC,IAAI,KAAK,eAAe;YAChC,OAAO,CAAC,IAAI,KAAK,cAAc;YAC/B,OAAO,CAAC,IAAI,KAAK,aAAa,CACjC,CAAC;QACF,OAAO,MAAM,CACX,KAAK,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,gBAAgB,EACxD,QAAQ,EACR,aAAa,CACd,CAAC;IACJ,CAAC;IAED,0EAA0E;IAC1E,MAAM,CAAC,eAAe,EAAE,kBAAkB,CAAC,GAAG,YAAY,CAAC,CAAC,CAAE,CAAC;IAC/D,MAAM,MAAM,GAAG,UAAU,CACvB,OAAQ,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,YAAY,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,EACnD,kBAAkB,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,YAAY,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,CAC9D,CAAC;IAEF,IAAI,MAAM,CAAC,MAAM,KAAK,IAAI,IAAI,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,MAAM,CAAC,GAAG,kBAAkB,EAAE,CAAC;QAC3E,QAAQ,CAAC,IAAI,CAAC;YACZ,IAAI,EAAE,UAAU;YAChB,MAAM,EACJ,aAAa,MAAM,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC,CAAC,oBAAoB,kBAAkB,gCAAgC;gBAC3G,QAAQ,eAAe,yBAAyB,CAAC,aAAa,CAAC,eAAe,CAAC,CAAC,WAAW,GAAG,GAAG,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,iBAAiB;SACjI,CAAC,CAAC;IACL,CAAC;IAED,4EAA4E;IAC5E,IAAI,QAAQ,CAAC,MAAM,KAAK,CAAC,IAAI,OAAO,CAAC,mBAAmB,KAAK,IAAI,EAAE,CAAC;QAClE,QAAQ,CAAC,IAAI,CAAC;YACZ,IAAI,EAAE,qBAAqB;YAC3B,MAAM,EACJ,uHAAuH;gBACvH,0GAA0G;gBAC1G,iEAAiE;SACpE,CAAC,CAAC;IACL,CAAC;IAED,IAAI,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACxB,wEAAwE;QACxE,qEAAqE;QACrE,uEAAuE;QACvE,iEAAiE;QACjE,MAAM,gBAAgB,GACpB,OAAO,CAAC,mBAAmB,KAAK,IAAI;YACpC,QAAQ,CAAC,CAAC,CAAE,CAAC,IAAI,KAAK,qBAAqB,CAAC;QAC9C,OAAO;YACL,OAAO,EAAE,WAAW;YACpB,QAAQ;YACR,aAAa;YACb,MAAM;YACN,OAAO,EAAE,GAAG,gBAAgB,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,WAAW,KAAK,QAAQ,CAAC,CAAC,CAAE,CAAC,MAAM,KAAK,QAAQ,CAAC,MAAM,kBAAkB;SACxH,CAAC;IACJ,CAAC;IAED,MAAM,SAAS,GAAG,MAAM,CAAC,UAAU,GAAG,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC;IAC3D,MAAM,SAAS,GACb,MAAM,CAAC,MAAM,KAAK,IAAI;QACpB,CAAC,CAAC,kCAAkC;QACpC,CAAC,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC;IACnD,OAAO;QACL,OAAO,EAAE,MAAM,CAAC,UAAU,GAAG,CAAC,CAAC,CAAC,CAAC,QAAQ,CAAC,CAAC,CAAC,YAAY;QACxD,QAAQ;QACR,aAAa;QACb,MAAM;QACN,OAAO,EACL,GAAG,MAAM,CAAC,UAAU,GAAG,CAAC,CAAC,CAAC,CAAC,QAAQ,CAAC,CAAC,CAAC,YAAY,SAAS,eAAe,SAAS;YACnF,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,UAAU,CAAC,IAAI,SAAS,iCAAiC,SAAS,GAAG;KAC3F,CAAC;AACJ,CAAC;AAED,oFAAoF;AACpF,SAAS,MAAM,CACb,OAAgB,EAChB,QAAgC,EAChC,aAAoD;IAEpD,OAAO;QACL,OAAO;QACP,QAAQ;QACR,aAAa;QACb,MAAM,EAAE,IAAI;QACZ,OAAO,EAAE,GAAG,QAAQ,CAAC,OAAO,CAAC,KAAK,QAAQ,CAAC,CAAC,CAAE,CAAC,MAAM,KAAK,QAAQ,CAAC,MAAM,kBAAkB;KAC5F,CAAC;AACJ,CAAC;AAED,MAAM,QAAQ,GAAsC;IAClD,MAAM,EAAE,QAAQ;IAChB,UAAU,EAAE,YAAY;IACxB,WAAW,EAAE,WAAW;IACxB,KAAK,EAAE,OAAO;IACd,gBAAgB,EAAE,gBAAgB;IAClC,OAAO,EAAE,SAAS;CACnB,CAAC"}
|
package/dist/ab-cli.d.ts
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `skillstate ab` — drive an A/B experiment and print a gated verdict.
|
|
3
|
+
*
|
|
4
|
+
* @non-paper — measurement infrastructure, not the paper's evaluation.
|
|
5
|
+
*
|
|
6
|
+
* Importing this module is side-effect free: it runs only when this file is
|
|
7
|
+
* the process entry (`node dist/ab-cli.js`), never on library import.
|
|
8
|
+
*
|
|
9
|
+
* ── The one rule this CLI enforces ───────────────────────────────────────
|
|
10
|
+
*
|
|
11
|
+
* It prints a percentage only when every gate passed. On any refusal it
|
|
12
|
+
* prints the gates instead. This is the whole point of the tool: the previous
|
|
13
|
+
* experiment's number was not wrong by arithmetic, it was wrong because
|
|
14
|
+
* nothing asked whether the instrumented arm had done anything.
|
|
15
|
+
*/
|
|
16
|
+
import type { ExperimentOptions } from './ab/verdict.js';
|
|
17
|
+
/** Options for {@link main}. */
|
|
18
|
+
export interface AbOptions extends ExperimentOptions {
|
|
19
|
+
/** Run files to read, in any order. */
|
|
20
|
+
readonly files: readonly string[];
|
|
21
|
+
/** Mark the task as defined over the historical trajectory. */
|
|
22
|
+
readonly taskNeedsTranscript?: boolean;
|
|
23
|
+
/** Overrides the value inferred from the run files. */
|
|
24
|
+
readonly minTrials?: number;
|
|
25
|
+
}
|
|
26
|
+
/**
|
|
27
|
+
* Read run files, run the gates, and print the report.
|
|
28
|
+
*
|
|
29
|
+
* Returns a process exit code. A refusal exits non-zero so a CI job wired to
|
|
30
|
+
* this cannot record an invalid experiment as a passing one.
|
|
31
|
+
*/
|
|
32
|
+
export declare function main(options: AbOptions): number;
|
|
33
|
+
/** Parse argv and run. Exported so tests can drive it without a subprocess. */
|
|
34
|
+
export declare function mainFromArgv(argv: readonly string[]): number;
|
|
35
|
+
//# sourceMappingURL=ab-cli.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"ab-cli.d.ts","sourceRoot":"","sources":["../src/ab-cli.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AAWH,OAAO,KAAK,EAAE,iBAAiB,EAAE,MAAM,iBAAiB,CAAC;AA0EzD,gCAAgC;AAChC,MAAM,WAAW,SAAU,SAAQ,iBAAiB;IAClD,uCAAuC;IACvC,QAAQ,CAAC,KAAK,EAAE,SAAS,MAAM,EAAE,CAAC;IAClC,+DAA+D;IAC/D,QAAQ,CAAC,mBAAmB,CAAC,EAAE,OAAO,CAAC;IACvC,uDAAuD;IACvD,QAAQ,CAAC,SAAS,CAAC,EAAE,MAAM,CAAC;CAC7B;AAED;;;;;GAKG;AACH,wBAAgB,IAAI,CAAC,OAAO,EAAE,SAAS,GAAG,MAAM,CAkC/C;AAED,+EAA+E;AAC/E,wBAAgB,YAAY,CAAC,IAAI,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,CAkD5D"}
|
package/dist/ab-cli.js
ADDED
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `skillstate ab` — drive an A/B experiment and print a gated verdict.
|
|
3
|
+
*
|
|
4
|
+
* @non-paper — measurement infrastructure, not the paper's evaluation.
|
|
5
|
+
*
|
|
6
|
+
* Importing this module is side-effect free: it runs only when this file is
|
|
7
|
+
* the process entry (`node dist/ab-cli.js`), never on library import.
|
|
8
|
+
*
|
|
9
|
+
* ── The one rule this CLI enforces ───────────────────────────────────────
|
|
10
|
+
*
|
|
11
|
+
* It prints a percentage only when every gate passed. On any refusal it
|
|
12
|
+
* prints the gates instead. This is the whole point of the tool: the previous
|
|
13
|
+
* experiment's number was not wrong by arithmetic, it was wrong because
|
|
14
|
+
* nothing asked whether the instrumented arm had done anything.
|
|
15
|
+
*/
|
|
16
|
+
import * as fs from 'node:fs';
|
|
17
|
+
import * as path from 'node:path';
|
|
18
|
+
import { pathToFileURL } from 'node:url';
|
|
19
|
+
import { assessEngagement, isWitnessed, withRejections } from './ab/engagement.js';
|
|
20
|
+
import { buildArmRecord } from './ab/record.js';
|
|
21
|
+
import { formatArmTable, formatVerdict } from './ab/report.js';
|
|
22
|
+
import { runExperiment } from './ab/verdict.js';
|
|
23
|
+
function parseRunFiles(files) {
|
|
24
|
+
const parsed = [];
|
|
25
|
+
for (const file of files) {
|
|
26
|
+
const raw = JSON.parse(fs.readFileSync(file, 'utf-8'));
|
|
27
|
+
for (const entry of Array.isArray(raw) ? raw : [raw]) {
|
|
28
|
+
parsed.push(entry);
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
return parsed;
|
|
32
|
+
}
|
|
33
|
+
function toRunRecord(trial) {
|
|
34
|
+
const report = assessEngagement(trial.stateSamples);
|
|
35
|
+
const witnessed = isWitnessed(trial.stateSamples);
|
|
36
|
+
// An unwitnessed trial is NOT recorded as inert. Nobody watched the file,
|
|
37
|
+
// and calling that "the integration did nothing" would be an accusation
|
|
38
|
+
// the harness cannot support. It is recorded as unwitnessed, which the
|
|
39
|
+
// sample-size and engagement gates then treat as a refusal.
|
|
40
|
+
const engagement = witnessed
|
|
41
|
+
? withRejections(report.evidence, trial.rejections ?? 0, trial.firstRejection)
|
|
42
|
+
: {
|
|
43
|
+
...report.evidence,
|
|
44
|
+
engaged: false,
|
|
45
|
+
writes: 0,
|
|
46
|
+
rejections: 0,
|
|
47
|
+
};
|
|
48
|
+
return {
|
|
49
|
+
arm: trial.arm,
|
|
50
|
+
trial: trial.trial,
|
|
51
|
+
task: trial.task,
|
|
52
|
+
model: trial.model,
|
|
53
|
+
hostVersion: trial.hostVersion,
|
|
54
|
+
sessionID: trial.sessionID,
|
|
55
|
+
usage: trial.usage,
|
|
56
|
+
engagement,
|
|
57
|
+
work: {
|
|
58
|
+
artifactDigest: trial.artifactDigest,
|
|
59
|
+
turns: trial.turns,
|
|
60
|
+
toolCalls: trial.toolCalls,
|
|
61
|
+
},
|
|
62
|
+
durationMs: trial.durationMs,
|
|
63
|
+
completed: trial.completed,
|
|
64
|
+
...(trial.correct === undefined ? {} : { outcome: { correct: trial.correct } }),
|
|
65
|
+
...(trial.error === undefined ? {} : { error: trial.error }),
|
|
66
|
+
};
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
69
|
+
* Read run files, run the gates, and print the report.
|
|
70
|
+
*
|
|
71
|
+
* Returns a process exit code. A refusal exits non-zero so a CI job wired to
|
|
72
|
+
* this cannot record an invalid experiment as a passing one.
|
|
73
|
+
*/
|
|
74
|
+
export function main(options) {
|
|
75
|
+
const trials = parseRunFiles(options.files);
|
|
76
|
+
const byArm = new Map();
|
|
77
|
+
for (const trial of trials) {
|
|
78
|
+
const runs = byArm.get(trial.arm) ?? [];
|
|
79
|
+
runs.push(toRunRecord(trial));
|
|
80
|
+
byArm.set(trial.arm, runs);
|
|
81
|
+
}
|
|
82
|
+
const arms = new Map();
|
|
83
|
+
const problems = [];
|
|
84
|
+
for (const [armId, runs] of byArm) {
|
|
85
|
+
const built = buildArmRecord(armId, runs);
|
|
86
|
+
if (!built.ok) {
|
|
87
|
+
problems.push(built.reason);
|
|
88
|
+
continue;
|
|
89
|
+
}
|
|
90
|
+
arms.set(armId, built.record);
|
|
91
|
+
}
|
|
92
|
+
const result = runExperiment(arms, {
|
|
93
|
+
...(options.taskNeedsTranscript === undefined
|
|
94
|
+
? {}
|
|
95
|
+
: { taskNeedsTranscript: options.taskNeedsTranscript }),
|
|
96
|
+
...(options.minTrials === undefined ? {} : { minTrials: options.minTrials }),
|
|
97
|
+
});
|
|
98
|
+
console.log(formatVerdict(result, formatArmTable(arms)));
|
|
99
|
+
for (const problem of problems) {
|
|
100
|
+
console.log(` - [input] ${problem}`);
|
|
101
|
+
}
|
|
102
|
+
const measured = result.verdict === 'saving' || result.verdict === 'regression';
|
|
103
|
+
return measured && problems.length === 0 ? 0 : 1;
|
|
104
|
+
}
|
|
105
|
+
/** Parse argv and run. Exported so tests can drive it without a subprocess. */
|
|
106
|
+
export function mainFromArgv(argv) {
|
|
107
|
+
const files = [];
|
|
108
|
+
let taskNeedsTranscript = false;
|
|
109
|
+
let minTrials;
|
|
110
|
+
for (let i = 0; i < argv.length; i += 1) {
|
|
111
|
+
const arg = argv[i];
|
|
112
|
+
if (arg === '--transcript-task') {
|
|
113
|
+
taskNeedsTranscript = true;
|
|
114
|
+
}
|
|
115
|
+
else if (arg === '--min-trials') {
|
|
116
|
+
const value = Number(argv[++i]);
|
|
117
|
+
if (!Number.isInteger(value) || value < 1) {
|
|
118
|
+
console.log('--min-trials needs a positive integer');
|
|
119
|
+
return 2;
|
|
120
|
+
}
|
|
121
|
+
minTrials = value;
|
|
122
|
+
}
|
|
123
|
+
else if (arg === '--help' || arg === '-h') {
|
|
124
|
+
console.log('usage: skillstate ab [--transcript-task] [--min-trials N] <run.json>...\n\n' +
|
|
125
|
+
'Each run file is a JSON object (or array of them) describing one trial:\n' +
|
|
126
|
+
' arm, trial, task, model, hostVersion, sessionID, usage{input,cacheRead,\n' +
|
|
127
|
+
' cacheWrite,output}, turns, toolCalls, artifactDigest, durationMs,\n' +
|
|
128
|
+
' completed, and stateSamples[{trial,step,content,fromSink?}].\n\n' +
|
|
129
|
+
'Exits 0 only when every gate passed and a saving or regression was found.');
|
|
130
|
+
return 0;
|
|
131
|
+
}
|
|
132
|
+
else if (arg.startsWith('-')) {
|
|
133
|
+
console.log(`unknown flag: ${arg}`);
|
|
134
|
+
return 2;
|
|
135
|
+
}
|
|
136
|
+
else {
|
|
137
|
+
files.push(arg);
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
if (files.length === 0) {
|
|
141
|
+
console.log('no run files given — nothing to compare');
|
|
142
|
+
return 2;
|
|
143
|
+
}
|
|
144
|
+
for (const file of files) {
|
|
145
|
+
if (!fs.existsSync(path.resolve(file))) {
|
|
146
|
+
console.log(`no such run file: ${file}`);
|
|
147
|
+
return 2;
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
return main({
|
|
151
|
+
files,
|
|
152
|
+
...(taskNeedsTranscript ? { taskNeedsTranscript: true } : {}),
|
|
153
|
+
...(minTrials === undefined ? {} : { minTrials }),
|
|
154
|
+
});
|
|
155
|
+
}
|
|
156
|
+
const isEntry = process.argv[1] !== undefined &&
|
|
157
|
+
pathToFileURL(process.argv[1]).href === import.meta.url;
|
|
158
|
+
if (isEntry) {
|
|
159
|
+
process.exitCode = mainFromArgv(process.argv.slice(2));
|
|
160
|
+
}
|
|
161
|
+
//# sourceMappingURL=ab-cli.js.map
|