@skillstate/bench 3.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -2
- package/dist/ab/engagement.d.ts +70 -0
- package/dist/ab/engagement.d.ts.map +1 -0
- package/dist/ab/engagement.js +102 -0
- package/dist/ab/engagement.js.map +1 -0
- package/dist/ab/index.d.ts +28 -0
- package/dist/ab/index.d.ts.map +1 -0
- package/dist/ab/index.js +20 -0
- package/dist/ab/index.js.map +1 -0
- package/dist/ab/opencode-usage.d.ts +52 -0
- package/dist/ab/opencode-usage.d.ts.map +1 -0
- package/dist/ab/opencode-usage.js +85 -0
- package/dist/ab/opencode-usage.js.map +1 -0
- package/dist/ab/record.d.ts +173 -0
- package/dist/ab/record.d.ts.map +1 -0
- package/dist/ab/record.js +98 -0
- package/dist/ab/record.js.map +1 -0
- package/dist/ab/replay.d.ts +193 -0
- package/dist/ab/replay.d.ts.map +1 -0
- package/dist/ab/replay.js +149 -0
- package/dist/ab/replay.js.map +1 -0
- package/dist/ab/report.d.ts +26 -0
- package/dist/ab/report.d.ts.map +1 -0
- package/dist/ab/report.js +98 -0
- package/dist/ab/report.js.map +1 -0
- package/dist/ab/stats-core.d.ts +84 -0
- package/dist/ab/stats-core.d.ts.map +1 -0
- package/dist/ab/stats-core.js +93 -0
- package/dist/ab/stats-core.js.map +1 -0
- package/dist/ab/survey.d.ts +131 -0
- package/dist/ab/survey.d.ts.map +1 -0
- package/dist/ab/survey.js +143 -0
- package/dist/ab/survey.js.map +1 -0
- package/dist/ab/usage.d.ts +99 -0
- package/dist/ab/usage.d.ts.map +1 -0
- package/dist/ab/usage.js +98 -0
- package/dist/ab/usage.js.map +1 -0
- package/dist/ab/verdict.d.ts +88 -0
- package/dist/ab/verdict.d.ts.map +1 -0
- package/dist/ab/verdict.js +251 -0
- package/dist/ab/verdict.js.map +1 -0
- package/dist/ab-cli.d.ts +35 -0
- package/dist/ab-cli.d.ts.map +1 -0
- package/dist/ab-cli.js +161 -0
- package/dist/ab-cli.js.map +1 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/package.json +5 -1
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The shape of one A/B run record.
|
|
3
|
+
*
|
|
4
|
+
* ── Why a record and not a number ────────────────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* The previous A/B produced a number — "39% saved" — from a run in which
|
|
7
|
+
* the instrumented arm never once touched the state file. The number was
|
|
8
|
+
* arithmetically correct and scientifically meaningless, and nothing in the
|
|
9
|
+
* pipeline could tell the difference, because the pipeline carried a number
|
|
10
|
+
* instead of a record.
|
|
11
|
+
*
|
|
12
|
+
* A record keeps the things that decide whether a comparison means anything:
|
|
13
|
+
* what the task was, what each arm spent, and — the part that was missing —
|
|
14
|
+
* **whether the instrumentation was engaged at all**. Gates in `verdict.ts`
|
|
15
|
+
* read these fields; none of them is derivable from token counts.
|
|
16
|
+
*
|
|
17
|
+
* @non-paper — measurement infrastructure for OUR host integration. Not
|
|
18
|
+
* part of the paper's own evaluation.
|
|
19
|
+
*/
|
|
20
|
+
/** Every arm, in the order an experiment should run them. */
|
|
21
|
+
export const ARM_IDS = ['plain', 'notes', 'paper'];
|
|
22
|
+
/**
|
|
23
|
+
* Whether an arm has the skillstate integration switched ON.
|
|
24
|
+
*
|
|
25
|
+
* `plain` is the control: no plugin, no state, no system fragment. The other
|
|
26
|
+
* two carry the integration in its two modes.
|
|
27
|
+
*/
|
|
28
|
+
export function isInstrumented(arm) {
|
|
29
|
+
return arm !== 'plain';
|
|
30
|
+
}
|
|
31
|
+
/**
|
|
32
|
+
* Tokens that constitute PROMPT cost.
|
|
33
|
+
*
|
|
34
|
+
* `input + cacheRead` is what the model had to be shown, whether it was
|
|
35
|
+
* billed as fresh or served from cache. `cacheWrite` is excluded because it
|
|
36
|
+
* is an artefact of a cold cache rather than of prompt size, and folding it
|
|
37
|
+
* in would make a run that happened to miss the cache look like an expensive
|
|
38
|
+
* one. `output` is excluded because the paper's claim is about prompt
|
|
39
|
+
* economy, not generation economy.
|
|
40
|
+
*/
|
|
41
|
+
export function promptTokens(usage) {
|
|
42
|
+
return usage.input + usage.cacheRead;
|
|
43
|
+
}
|
|
44
|
+
/** Engagement for a run whose arm carries no integration. */
|
|
45
|
+
export const CONTROL_ENGAGEMENT = {
|
|
46
|
+
engaged: false,
|
|
47
|
+
writes: 0,
|
|
48
|
+
sinkWrites: 0,
|
|
49
|
+
rejections: 0,
|
|
50
|
+
hadStateAtStep0: false,
|
|
51
|
+
};
|
|
52
|
+
/**
|
|
53
|
+
* Build an {@link ArmRecord} from its runs, checking they are commensurable.
|
|
54
|
+
*
|
|
55
|
+
* A trial index must be unique within an arm: two runs labelled `trial: 0`
|
|
56
|
+
* are two samples of the same cell, and silently averaging them would hide
|
|
57
|
+
* exactly the variance the harness exists to surface.
|
|
58
|
+
*
|
|
59
|
+
* Returns a typed failure rather than throwing — a malformed arm is a
|
|
60
|
+
* harness bug, and the caller must be able to report it as one.
|
|
61
|
+
*/
|
|
62
|
+
export function buildArmRecord(arm, runs) {
|
|
63
|
+
if (runs.length === 0) {
|
|
64
|
+
return { ok: false, reason: `arm ${arm} has no runs` };
|
|
65
|
+
}
|
|
66
|
+
const first = runs[0];
|
|
67
|
+
const mismatched = (field) => {
|
|
68
|
+
for (const run of runs) {
|
|
69
|
+
if (run[field] !== first[field]) {
|
|
70
|
+
return `arm ${arm}: ${String(field)} differs across runs ("${String(first[field])}" vs "${String(run[field])}")`;
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
return undefined;
|
|
74
|
+
};
|
|
75
|
+
for (const field of ['task', 'model', 'hostVersion', 'arm']) {
|
|
76
|
+
const problem = mismatched(field);
|
|
77
|
+
if (problem !== undefined)
|
|
78
|
+
return { ok: false, reason: problem };
|
|
79
|
+
}
|
|
80
|
+
const seen = new Set();
|
|
81
|
+
for (const run of runs) {
|
|
82
|
+
if (seen.has(run.trial)) {
|
|
83
|
+
return { ok: false, reason: `arm ${arm}: duplicate trial index ${run.trial}` };
|
|
84
|
+
}
|
|
85
|
+
seen.add(run.trial);
|
|
86
|
+
}
|
|
87
|
+
return {
|
|
88
|
+
ok: true,
|
|
89
|
+
record: {
|
|
90
|
+
arm,
|
|
91
|
+
task: first.task,
|
|
92
|
+
model: first.model,
|
|
93
|
+
hostVersion: first.hostVersion,
|
|
94
|
+
runs: [...runs].sort((a, b) => a.trial - b.trial),
|
|
95
|
+
},
|
|
96
|
+
};
|
|
97
|
+
}
|
|
98
|
+
//# sourceMappingURL=record.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"record.js","sourceRoot":"","sources":["../../src/ab/record.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;GAkBG;AAKH,6DAA6D;AAC7D,MAAM,CAAC,MAAM,OAAO,GAAqB,CAAC,OAAO,EAAE,OAAO,EAAE,OAAO,CAAC,CAAC;AAErE;;;;;GAKG;AACH,MAAM,UAAU,cAAc,CAAC,GAAU;IACvC,OAAO,GAAG,KAAK,OAAO,CAAC;AACzB,CAAC;AAcD;;;;;;;;;GASG;AACH,MAAM,UAAU,YAAY,CAAC,KAAiB;IAC5C,OAAO,KAAK,CAAC,KAAK,GAAG,KAAK,CAAC,SAAS,CAAC;AACvC,CAAC;AA8BD,6DAA6D;AAC7D,MAAM,CAAC,MAAM,kBAAkB,GAAuB;IACpD,OAAO,EAAE,KAAK;IACd,MAAM,EAAE,CAAC;IACT,UAAU,EAAE,CAAC;IACb,UAAU,EAAE,CAAC;IACb,eAAe,EAAE,KAAK;CACvB,CAAC;AAgFF;;;;;;;;;GASG;AACH,MAAM,UAAU,cAAc,CAC5B,GAAU,EACV,IAA0B;IAE1B,IAAI,IAAI,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACtB,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,MAAM,EAAE,OAAO,GAAG,cAAc,EAAE,CAAC;IACzD,CAAC;IACD,MAAM,KAAK,GAAG,IAAI,CAAC,CAAC,CAAE,CAAC;IACvB,MAAM,UAAU,GAAG,CAAC,KAAsB,EAAsB,EAAE;QAChE,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;YACvB,IAAI,GAAG,CAAC,KAAK,CAAC,KAAK,KAAK,CAAC,KAAK,CAAC,EAAE,CAAC;gBAChC,OAAO,OAAO,GAAG,KAAK,MAAM,CAAC,KAAK,CAAC,0BAA0B,MAAM,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,SAAS,MAAM,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC;YACnH,CAAC;QACH,CAAC;QACD,OAAO,SAAS,CAAC;IACnB,CAAC,CAAC;IACF,KAAK,MAAM,KAAK,IAAI,CAAC,MAAM,EAAE,OAAO,EAAE,aAAa,EAAE,KAAK,CAAU,EAAE,CAAC;QACrE,MAAM,OAAO,GAAG,UAAU,CAAC,KAAK,CAAC,CAAC;QAClC,IAAI,OAAO,KAAK,SAAS;YAAE,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,MAAM,EAAE,OAAO,EAAE,CAAC;IACnE,CAAC;IAED,MAAM,IAAI,GAAG,IAAI,GAAG,EAAU,CAAC;IAC/B,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;QACvB,IAAI,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,KAAK,CAAC,EAAE,CAAC;YACxB,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,MAAM,EAAE,OAAO,GAAG,2BAA2B,GAAG,CAAC,KAAK,EAAE,EAAE,CAAC;QACjF,CAAC;QACD,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC;IACtB,CAAC;IAED,OAAO;QACL,EAAE,EAAE,IAAI;QACR,MAAM,EAAE;YACN,GAAG;YACH,IAAI,EAAE,KAAK,CAAC,IAAI;YAChB,KAAK,EAAE,KAAK,CAAC,KAAK;YAClB,WAAW,EAAE,KAAK,CAAC,WAAW;YAC9B,IAAI,EAAE,CAAC,GAAG,IAAI,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,CAAC;SAClD;KACF,CAAC;AACJ,CAAC"}
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Reconstructing the host's token curve, and what a bounded prompt would cost.
|
|
3
|
+
*
|
|
4
|
+
* ── Why this exists, given the A/B harness ───────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* The harness answers "does the integration save tokens, on a live host". That
|
|
7
|
+
* question needs a live host, and the account quota is currently negative, so
|
|
8
|
+
* it cannot be answered today by running more experiments.
|
|
9
|
+
*
|
|
10
|
+
* This module answers a different and much narrower question that needs no
|
|
11
|
+
* model at all: **given a real session's real token accounting, what would a
|
|
12
|
+
* bounded Aₜ = (P, Σₜ, Oₜ) prompt have cost on those same steps?**
|
|
13
|
+
*
|
|
14
|
+
* That is not a substitute for the A/B, and the difference matters:
|
|
15
|
+
*
|
|
16
|
+
* - the A/B measures whether an agent, given a bounded prompt, still does the
|
|
17
|
+
* work — an outcome question this cannot touch;
|
|
18
|
+
* - this measures the COST side only, using token counts the host itself
|
|
19
|
+
* recorded, with no model in the loop.
|
|
20
|
+
*
|
|
21
|
+
* Both are needed. A cost win with no task completion is worthless, and a task
|
|
22
|
+
* completion with no cost win is not the claim the paper makes. This module
|
|
23
|
+
* deliberately reports the half it can actually measure, and refuses to
|
|
24
|
+
* extrapolate to the half it cannot.
|
|
25
|
+
*
|
|
26
|
+
* ── Where the numbers come from ──────────────────────────────────────────
|
|
27
|
+
*
|
|
28
|
+
* OpenCode records per-message `tokens` — `input`, `cache.read`,
|
|
29
|
+
* `cache.write`, `output`, `reasoning` — in its own store. Those are the
|
|
30
|
+
* host's own accounting for a real session, not our reconstruction of it.
|
|
31
|
+
*
|
|
32
|
+
* `cache.read` is the interesting one, and the reason this is not a
|
|
33
|
+
* hypothetical. A growing transcript does not merely cost more to send: the
|
|
34
|
+
* host caches the prefix, so the re-sent history is billed as cache reads.
|
|
35
|
+
* Ignoring them would report a saving of a few percent and hide the actual
|
|
36
|
+
* shape of the cost.
|
|
37
|
+
*
|
|
38
|
+
* @non-paper — measurement infrastructure, not the paper's evaluation.
|
|
39
|
+
*/
|
|
40
|
+
import type { Distribution, EffectSize } from './stats-core.js';
|
|
41
|
+
/** One assistant step's token accounting, as the host recorded it. */
|
|
42
|
+
export interface StepUsage {
|
|
43
|
+
/** Fresh, uncached input tokens. */
|
|
44
|
+
readonly input: number;
|
|
45
|
+
/** Input tokens served from the prefix cache. */
|
|
46
|
+
readonly cacheRead: number;
|
|
47
|
+
/** Output tokens. */
|
|
48
|
+
readonly output: number;
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* What a cache read is worth relative to a fresh input token.
|
|
52
|
+
*
|
|
53
|
+
* `1/10` is the usual published rate for prompt caching, and it is a DEFAULT,
|
|
54
|
+
* not a measurement: the real multiplier is the provider's, and this repo has
|
|
55
|
+
* no access to anyone's invoice. The number exists because adding `cache.read`
|
|
56
|
+
* to `input` at face value is the single easiest way to overstate a token
|
|
57
|
+
* saving, and that is worth a knob rather than a silent assumption.
|
|
58
|
+
*
|
|
59
|
+
* A cache read is NOT free: the model is shown the same text either way, and
|
|
60
|
+
* the paper's claim is about what the model is exposed to. The discount
|
|
61
|
+
* reflects PRICE, not attention.
|
|
62
|
+
*/
|
|
63
|
+
export declare const DEFAULT_CACHE_READ_DISCOUNT = 0.1;
|
|
64
|
+
/** Fresh input plus cache reads, i.e. everything the model was shown. */
|
|
65
|
+
export declare function promptTokensOf(step: StepUsage): number;
|
|
66
|
+
/** Options for {@link reconstruct}. */
|
|
67
|
+
export interface ReconstructOptions {
|
|
68
|
+
/**
|
|
69
|
+
* Tokens a single bounded Aₜ prompt costs.
|
|
70
|
+
*
|
|
71
|
+
* Defaults to `DEFAULT_BOUNDED_PROMPT_TOKENS`, which is the measured size of
|
|
72
|
+
* a real A.4 prompt for a populated state — not an optimistic guess. It is
|
|
73
|
+
* overridable because the honest comparison depends on the spec and the
|
|
74
|
+
* state: a procedure with twenty schema fields costs more per step than one
|
|
75
|
+
* with two, and pretending otherwise would flatter the integration.
|
|
76
|
+
*/
|
|
77
|
+
readonly boundedPromptTokens?: number;
|
|
78
|
+
/**
|
|
79
|
+
* Tokens of Σₜ, the state block inside the bounded prompt.
|
|
80
|
+
*
|
|
81
|
+
* Σₜ is not free — it grows with what the agent records — so a paper prompt
|
|
82
|
+
* is not literally constant. Including it separately, rather than folding it
|
|
83
|
+
* into the constant, keeps that growth visible instead of hidden inside a
|
|
84
|
+
* number that looks flat.
|
|
85
|
+
*/
|
|
86
|
+
readonly stateTokens?: number;
|
|
87
|
+
/**
|
|
88
|
+
* What a cache read costs relative to a fresh input token.
|
|
89
|
+
*
|
|
90
|
+
* Defaults to {@link DEFAULT_CACHE_READ_DISCOUNT}. Overridable because the
|
|
91
|
+
* real multiplier is the provider's, and a number nobody can change is a
|
|
92
|
+
* number nobody should trust.
|
|
93
|
+
*/
|
|
94
|
+
readonly cacheReadDiscount?: number;
|
|
95
|
+
}
|
|
96
|
+
/**
|
|
97
|
+
* Measured A.4 prompt size for a populated state, in tokens.
|
|
98
|
+
*
|
|
99
|
+
* 1 800 is the paper's own Table 1 figure and is close to what
|
|
100
|
+
* `PromptTransformer.formatPaper` produces for a realistic spec: instructions
|
|
101
|
+
* plus a compact state block plus a short observation. It is a DEFAULT, and
|
|
102
|
+
* `reconstruct` takes an override, precisely so nobody quotes it as our
|
|
103
|
+
* measurement of anything.
|
|
104
|
+
*/
|
|
105
|
+
export declare const DEFAULT_BOUNDED_PROMPT_TOKENS = 1800;
|
|
106
|
+
/** A real session's per-step token accounting. */
|
|
107
|
+
export interface HostSession {
|
|
108
|
+
readonly sessionID: string;
|
|
109
|
+
/** Title or task description, for the report. */
|
|
110
|
+
readonly label: string;
|
|
111
|
+
/** Ordered per-step usage, exactly as the host recorded it. */
|
|
112
|
+
readonly steps: readonly StepUsage[];
|
|
113
|
+
}
|
|
114
|
+
/** The two curves, and what they cost. */
|
|
115
|
+
export interface ReconstructResult {
|
|
116
|
+
readonly sessionID: string;
|
|
117
|
+
readonly label: string;
|
|
118
|
+
/** Number of steps replayed. */
|
|
119
|
+
readonly steps: number;
|
|
120
|
+
/** What the host actually spent on prompts, summed. */
|
|
121
|
+
readonly hostPromptTokens: number;
|
|
122
|
+
/** What bounded Aₜ prompts would have cost, summed. */
|
|
123
|
+
readonly boundedPromptTokens: number;
|
|
124
|
+
/** `host − bounded`, summed. Negative means the transcript was cheaper. */
|
|
125
|
+
readonly savedTokens: number;
|
|
126
|
+
/** `saved / host`, as a fraction. Null when the host spent nothing. */
|
|
127
|
+
readonly savedFraction: number | null;
|
|
128
|
+
/** Fresh, uncached input tokens the host charged for. */
|
|
129
|
+
readonly freshInputTokens: number;
|
|
130
|
+
/** Cache-read tokens the host charged for. */
|
|
131
|
+
readonly cacheReadTokens: number;
|
|
132
|
+
/**
|
|
133
|
+
* The session priced in input-equivalent tokens, with cache reads
|
|
134
|
+
* discounted.
|
|
135
|
+
*
|
|
136
|
+
* The honest denominator. `hostPromptTokens` adds a cache read to a fresh
|
|
137
|
+
* input as if they cost the same, which they do not, and a token saving
|
|
138
|
+
* quoted from that sum is inflated by roughly an order of magnitude on a
|
|
139
|
+
* cache-heavy corpus.
|
|
140
|
+
*/
|
|
141
|
+
readonly hostEffectiveTokens: number;
|
|
142
|
+
/**
|
|
143
|
+
* The saving priced properly: `(effective − bounded) / effective`.
|
|
144
|
+
*
|
|
145
|
+
* Always smaller than {@link savedFraction}, and the one to quote.
|
|
146
|
+
*/
|
|
147
|
+
readonly savedEffectiveFraction: number | null;
|
|
148
|
+
/** Host prompt tokens per step, in order. */
|
|
149
|
+
readonly hostPerStep: readonly number[];
|
|
150
|
+
/** Bounded prompt tokens per step, in order. */
|
|
151
|
+
readonly boundedPerStep: readonly number[];
|
|
152
|
+
/**
|
|
153
|
+
* How much the host's per-step prompt grew across the session.
|
|
154
|
+
*
|
|
155
|
+
* This is the O(T) signature: a bounded prompt has a near-zero slope, and a
|
|
156
|
+
* transcript's slope is the entire cost of the next hundred steps.
|
|
157
|
+
*/
|
|
158
|
+
readonly hostSlope: number;
|
|
159
|
+
/** The same slope for the bounded prompt. Expected to be 0. */
|
|
160
|
+
readonly boundedSlope: number;
|
|
161
|
+
/** The effect size in MADs, for comparison against the noise floor. */
|
|
162
|
+
readonly effect: EffectSize;
|
|
163
|
+
/** Host and bounded distributions, for the report. */
|
|
164
|
+
readonly distributions: {
|
|
165
|
+
readonly host: Distribution;
|
|
166
|
+
readonly bounded: Distribution;
|
|
167
|
+
};
|
|
168
|
+
}
|
|
169
|
+
/** Total prompt tokens for one step: fresh input plus cache reads. */
|
|
170
|
+
export declare function freshInputTokens(session: HostSession): number;
|
|
171
|
+
/** Total cache-read tokens for a session. */
|
|
172
|
+
export declare function cacheReadTokens(session: HostSession): number;
|
|
173
|
+
/**
|
|
174
|
+
* Compare a real session's transcript cost against a bounded prompt.
|
|
175
|
+
*
|
|
176
|
+
* Pure arithmetic over the host's own numbers. No model, no network, no clock
|
|
177
|
+
* — which is what makes it runnable when the quota is exhausted, and what
|
|
178
|
+
* makes it deterministic enough to assert on in a test.
|
|
179
|
+
*/
|
|
180
|
+
export declare function reconstruct(session: HostSession, options?: ReconstructOptions): ReconstructResult;
|
|
181
|
+
/**
|
|
182
|
+
* Whether a reconstruction is strong enough to act on.
|
|
183
|
+
*
|
|
184
|
+
* The same discipline as the A/B gates, applied to the offline path. A
|
|
185
|
+
* one-step session, or one whose transcript never grew, cannot distinguish
|
|
186
|
+
* the two curves — and a saving computed from a flat transcript is an artifact
|
|
187
|
+
* of the constant, not a result.
|
|
188
|
+
*/
|
|
189
|
+
export declare function assessReconstruction(result: ReconstructResult): {
|
|
190
|
+
readonly usable: boolean;
|
|
191
|
+
readonly reasons: readonly string[];
|
|
192
|
+
};
|
|
193
|
+
//# sourceMappingURL=replay.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"replay.d.ts","sourceRoot":"","sources":["../../src/ab/replay.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAsCG;AAGH,OAAO,KAAK,EAAE,YAAY,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAEhE,sEAAsE;AACtE,MAAM,WAAW,SAAS;IACxB,oCAAoC;IACpC,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,iDAAiD;IACjD,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,qBAAqB;IACrB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;CACzB;AAED;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,2BAA2B,MAAM,CAAC;AAE/C,yEAAyE;AACzE,wBAAgB,cAAc,CAAC,IAAI,EAAE,SAAS,GAAG,MAAM,CAEtD;AAED,uCAAuC;AACvC,MAAM,WAAW,kBAAkB;IACjC;;;;;;;;OAQG;IACH,QAAQ,CAAC,mBAAmB,CAAC,EAAE,MAAM,CAAC;IACtC;;;;;;;OAOG;IACH,QAAQ,CAAC,WAAW,CAAC,EAAE,MAAM,CAAC;IAC9B;;;;;;OAMG;IACH,QAAQ,CAAC,iBAAiB,CAAC,EAAE,MAAM,CAAC;CACrC;AAED;;;;;;;;GAQG;AACH,eAAO,MAAM,6BAA6B,OAAO,CAAC;AAElD,kDAAkD;AAClD,MAAM,WAAW,WAAW;IAC1B,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,iDAAiD;IACjD,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,+DAA+D;IAC/D,QAAQ,CAAC,KAAK,EAAE,SAAS,SAAS,EAAE,CAAC;CACtC;AAED,0CAA0C;AAC1C,MAAM,WAAW,iBAAiB;IAChC,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,gCAAgC;IAChC,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,uDAAuD;IACvD,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,uDAAuD;IACvD,QAAQ,CAAC,mBAAmB,EAAE,MAAM,CAAC;IACrC,2EAA2E;IAC3E,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAC7B,uEAAuE;IACvE,QAAQ,CAAC,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IACtC,yDAAyD;IACzD,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,8CAA8C;IAC9C,QAAQ,CAAC,eAAe,EAAE,MAAM,CAAC;IACjC;;;;;;;;OAQG;IACH,QAAQ,CAAC,mBAAmB,EAAE,MAAM,CAAC;IACrC;;;;OAIG;IACH,QAAQ,CAAC,sBAAsB,EAAE,MAAM,GAAG,IAAI,CAAC;IAC/C,6CAA6C;IAC7C,QAAQ,CAAC,WAAW,EAAE,SAAS,MAAM,EAAE,CAAC;IACxC,gDAAgD;IAChD,QAAQ,CAAC,cAAc,EAAE,SAAS,MAAM,EAAE,CAAC;IAC3C;;;;;OAKG;IACH,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,+DAA+D;IAC/D,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;IAC9B,uEAAuE;IACvE,QAAQ,CAAC,MAAM,EAAE,UAAU,CAAC;IAC5B,sDAAsD;IACtD,QAAQ,CAAC,aAAa,EAAE;QACtB,QAAQ,CAAC,IAAI,EAAE,YAAY,CAAC;QAC5B,QAAQ,CAAC,OAAO,EAAE,YAAY,CAAC;KAChC,CAAC;CACH;AAED,sEAAsE;AACtE,wBAAgB,gBAAgB,CAAC,OAAO,EAAE,WAAW,GAAG,MAAM,CAE7D;AAED,6CAA6C;AAC7C,wBAAgB,eAAe,CAAC,OAAO,EAAE,WAAW,GAAG,MAAM,CAE5D;AAED;;;;;;GAMG;AACH,wBAAgB,WAAW,CACzB,OAAO,EAAE,WAAW,EACpB,OAAO,GAAE,kBAAuB,GAC/B,iBAAiB,CA6CnB;AAED;;;;;;;GAOG;AACH,wBAAgB,oBAAoB,CAAC,MAAM,EAAE,iBAAiB,GAAG;IAC/D,QAAQ,CAAC,MAAM,EAAE,OAAO,CAAC;IACzB,QAAQ,CAAC,OAAO,EAAE,SAAS,MAAM,EAAE,CAAC;CACrC,CAiBA"}
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Reconstructing the host's token curve, and what a bounded prompt would cost.
|
|
3
|
+
*
|
|
4
|
+
* ── Why this exists, given the A/B harness ───────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* The harness answers "does the integration save tokens, on a live host". That
|
|
7
|
+
* question needs a live host, and the account quota is currently negative, so
|
|
8
|
+
* it cannot be answered today by running more experiments.
|
|
9
|
+
*
|
|
10
|
+
* This module answers a different and much narrower question that needs no
|
|
11
|
+
* model at all: **given a real session's real token accounting, what would a
|
|
12
|
+
* bounded Aₜ = (P, Σₜ, Oₜ) prompt have cost on those same steps?**
|
|
13
|
+
*
|
|
14
|
+
* That is not a substitute for the A/B, and the difference matters:
|
|
15
|
+
*
|
|
16
|
+
* - the A/B measures whether an agent, given a bounded prompt, still does the
|
|
17
|
+
* work — an outcome question this cannot touch;
|
|
18
|
+
* - this measures the COST side only, using token counts the host itself
|
|
19
|
+
* recorded, with no model in the loop.
|
|
20
|
+
*
|
|
21
|
+
* Both are needed. A cost win with no task completion is worthless, and a task
|
|
22
|
+
* completion with no cost win is not the claim the paper makes. This module
|
|
23
|
+
* deliberately reports the half it can actually measure, and refuses to
|
|
24
|
+
* extrapolate to the half it cannot.
|
|
25
|
+
*
|
|
26
|
+
* ── Where the numbers come from ──────────────────────────────────────────
|
|
27
|
+
*
|
|
28
|
+
* OpenCode records per-message `tokens` — `input`, `cache.read`,
|
|
29
|
+
* `cache.write`, `output`, `reasoning` — in its own store. Those are the
|
|
30
|
+
* host's own accounting for a real session, not our reconstruction of it.
|
|
31
|
+
*
|
|
32
|
+
* `cache.read` is the interesting one, and the reason this is not a
|
|
33
|
+
* hypothetical. A growing transcript does not merely cost more to send: the
|
|
34
|
+
* host caches the prefix, so the re-sent history is billed as cache reads.
|
|
35
|
+
* Ignoring them would report a saving of a few percent and hide the actual
|
|
36
|
+
* shape of the cost.
|
|
37
|
+
*
|
|
38
|
+
* @non-paper — measurement infrastructure, not the paper's evaluation.
|
|
39
|
+
*/
|
|
40
|
+
import { describe, effectSize, MIN_EFFECT_IN_MADS } from './stats-core.js';
|
|
41
|
+
/**
|
|
42
|
+
* What a cache read is worth relative to a fresh input token.
|
|
43
|
+
*
|
|
44
|
+
* `1/10` is the usual published rate for prompt caching, and it is a DEFAULT,
|
|
45
|
+
* not a measurement: the real multiplier is the provider's, and this repo has
|
|
46
|
+
* no access to anyone's invoice. The number exists because adding `cache.read`
|
|
47
|
+
* to `input` at face value is the single easiest way to overstate a token
|
|
48
|
+
* saving, and that is worth a knob rather than a silent assumption.
|
|
49
|
+
*
|
|
50
|
+
* A cache read is NOT free: the model is shown the same text either way, and
|
|
51
|
+
* the paper's claim is about what the model is exposed to. The discount
|
|
52
|
+
* reflects PRICE, not attention.
|
|
53
|
+
*/
|
|
54
|
+
export const DEFAULT_CACHE_READ_DISCOUNT = 0.1;
|
|
55
|
+
/** Fresh input plus cache reads, i.e. everything the model was shown. */
|
|
56
|
+
export function promptTokensOf(step) {
|
|
57
|
+
return step.input + step.cacheRead;
|
|
58
|
+
}
|
|
59
|
+
/**
|
|
60
|
+
* Measured A.4 prompt size for a populated state, in tokens.
|
|
61
|
+
*
|
|
62
|
+
* 1 800 is the paper's own Table 1 figure and is close to what
|
|
63
|
+
* `PromptTransformer.formatPaper` produces for a realistic spec: instructions
|
|
64
|
+
* plus a compact state block plus a short observation. It is a DEFAULT, and
|
|
65
|
+
* `reconstruct` takes an override, precisely so nobody quotes it as our
|
|
66
|
+
* measurement of anything.
|
|
67
|
+
*/
|
|
68
|
+
export const DEFAULT_BOUNDED_PROMPT_TOKENS = 1800;
|
|
69
|
+
/** Total prompt tokens for one step: fresh input plus cache reads. */
|
|
70
|
+
export function freshInputTokens(session) {
|
|
71
|
+
return session.steps.reduce((sum, step) => sum + step.input, 0);
|
|
72
|
+
}
|
|
73
|
+
/** Total cache-read tokens for a session. */
|
|
74
|
+
export function cacheReadTokens(session) {
|
|
75
|
+
return session.steps.reduce((sum, step) => sum + step.cacheRead, 0);
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* Compare a real session's transcript cost against a bounded prompt.
|
|
79
|
+
*
|
|
80
|
+
* Pure arithmetic over the host's own numbers. No model, no network, no clock
|
|
81
|
+
* — which is what makes it runnable when the quota is exhausted, and what
|
|
82
|
+
* makes it deterministic enough to assert on in a test.
|
|
83
|
+
*/
|
|
84
|
+
export function reconstruct(session, options = {}) {
|
|
85
|
+
const stateTokens = options.stateTokens ?? 0;
|
|
86
|
+
const perStepBounded = (options.boundedPromptTokens ?? DEFAULT_BOUNDED_PROMPT_TOKENS) + stateTokens;
|
|
87
|
+
const hostPerStep = session.steps.map(promptTokensOf);
|
|
88
|
+
const boundedPerStep = session.steps.map(() => perStepBounded);
|
|
89
|
+
const hostPromptTokens = hostPerStep.reduce((a, b) => a + b, 0);
|
|
90
|
+
const boundedPromptTokens = boundedPerStep.reduce((a, b) => a + b, 0);
|
|
91
|
+
const savedTokens = hostPromptTokens - boundedPromptTokens;
|
|
92
|
+
const fresh = freshInputTokens(session);
|
|
93
|
+
const cached = cacheReadTokens(session);
|
|
94
|
+
const discount = options.cacheReadDiscount ?? DEFAULT_CACHE_READ_DISCOUNT;
|
|
95
|
+
// What the session cost in INPUT-EQUIVALENT tokens: cache reads are counted
|
|
96
|
+
// at their price, not at face value. This is the number that survives a
|
|
97
|
+
// billing argument; the raw total does not.
|
|
98
|
+
const hostEffectiveTokens = fresh + cached * discount;
|
|
99
|
+
return {
|
|
100
|
+
sessionID: session.sessionID,
|
|
101
|
+
label: session.label,
|
|
102
|
+
steps: session.steps.length,
|
|
103
|
+
hostPromptTokens,
|
|
104
|
+
boundedPromptTokens,
|
|
105
|
+
savedTokens,
|
|
106
|
+
savedFraction: hostPromptTokens === 0 ? null : savedTokens / hostPromptTokens,
|
|
107
|
+
freshInputTokens: fresh,
|
|
108
|
+
cacheReadTokens: cached,
|
|
109
|
+
hostEffectiveTokens,
|
|
110
|
+
savedEffectiveFraction: hostEffectiveTokens === 0
|
|
111
|
+
? null
|
|
112
|
+
: (hostEffectiveTokens - boundedPromptTokens) / hostEffectiveTokens,
|
|
113
|
+
hostPerStep,
|
|
114
|
+
boundedPerStep,
|
|
115
|
+
// Slope across the whole session: last step minus first. A transcript's
|
|
116
|
+
// growth is the story; the total alone averages it away.
|
|
117
|
+
hostSlope: hostPerStep.length < 2 ? 0 : hostPerStep[hostPerStep.length - 1] - hostPerStep[0],
|
|
118
|
+
boundedSlope: boundedPerStep.length < 2 ? 0 : boundedPerStep[boundedPerStep.length - 1] - boundedPerStep[0],
|
|
119
|
+
// Signed so a positive effect means the transcript cost more.
|
|
120
|
+
effect: effectSize(hostPerStep, boundedPerStep),
|
|
121
|
+
distributions: {
|
|
122
|
+
host: describe(hostPerStep),
|
|
123
|
+
bounded: describe(boundedPerStep),
|
|
124
|
+
},
|
|
125
|
+
};
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* Whether a reconstruction is strong enough to act on.
|
|
129
|
+
*
|
|
130
|
+
* The same discipline as the A/B gates, applied to the offline path. A
|
|
131
|
+
* one-step session, or one whose transcript never grew, cannot distinguish
|
|
132
|
+
* the two curves — and a saving computed from a flat transcript is an artifact
|
|
133
|
+
* of the constant, not a result.
|
|
134
|
+
*/
|
|
135
|
+
export function assessReconstruction(result) {
|
|
136
|
+
const reasons = [];
|
|
137
|
+
if (result.steps < 2) {
|
|
138
|
+
reasons.push('a single step cannot show growth; need at least 2');
|
|
139
|
+
}
|
|
140
|
+
if (result.hostSlope <= 0) {
|
|
141
|
+
reasons.push('the transcript did not grow across the session, so there is no O(T) cost to remove; ' +
|
|
142
|
+
'this session is too short or too uniform to distinguish the two curves');
|
|
143
|
+
}
|
|
144
|
+
if (result.effect.inMads !== null && Math.abs(result.effect.inMads) < MIN_EFFECT_IN_MADS) {
|
|
145
|
+
reasons.push(`the gap is ${result.effect.inMads.toFixed(2)} MADs, under the ${MIN_EFFECT_IN_MADS} needed to call it a signal`);
|
|
146
|
+
}
|
|
147
|
+
return { usable: reasons.length === 0, reasons };
|
|
148
|
+
}
|
|
149
|
+
//# sourceMappingURL=replay.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"replay.js","sourceRoot":"","sources":["../../src/ab/replay.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAsCG;AAEH,OAAO,EAAE,QAAQ,EAAE,UAAU,EAAE,kBAAkB,EAAE,MAAM,iBAAiB,CAAC;AAa3E;;;;;;;;;;;;GAYG;AACH,MAAM,CAAC,MAAM,2BAA2B,GAAG,GAAG,CAAC;AAE/C,yEAAyE;AACzE,MAAM,UAAU,cAAc,CAAC,IAAe;IAC5C,OAAO,IAAI,CAAC,KAAK,GAAG,IAAI,CAAC,SAAS,CAAC;AACrC,CAAC;AAiCD;;;;;;;;GAQG;AACH,MAAM,CAAC,MAAM,6BAA6B,GAAG,IAAI,CAAC;AAmElD,sEAAsE;AACtE,MAAM,UAAU,gBAAgB,CAAC,OAAoB;IACnD,OAAO,OAAO,CAAC,KAAK,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,IAAI,EAAE,EAAE,CAAC,GAAG,GAAG,IAAI,CAAC,KAAK,EAAE,CAAC,CAAC,CAAC;AAClE,CAAC;AAED,6CAA6C;AAC7C,MAAM,UAAU,eAAe,CAAC,OAAoB;IAClD,OAAO,OAAO,CAAC,KAAK,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,IAAI,EAAE,EAAE,CAAC,GAAG,GAAG,IAAI,CAAC,SAAS,EAAE,CAAC,CAAC,CAAC;AACtE,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,WAAW,CACzB,OAAoB,EACpB,OAAO,GAAuB,EAAE;IAEhC,MAAM,WAAW,GAAG,OAAO,CAAC,WAAW,IAAI,CAAC,CAAC;IAC7C,MAAM,cAAc,GAAG,CAAC,OAAO,CAAC,mBAAmB,IAAI,6BAA6B,CAAC,GAAG,WAAW,CAAC;IAEpG,MAAM,WAAW,GAAG,OAAO,CAAC,KAAK,CAAC,GAAG,CAAC,cAAc,CAAC,CAAC;IACtD,MAAM,cAAc,GAAG,OAAO,CAAC,KAAK,CAAC,GAAG,CAAC,GAAG,EAAE,CAAC,cAAc,CAAC,CAAC;IAC/D,MAAM,gBAAgB,GAAG,WAAW,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,EAAE,CAAC,CAAC,CAAC;IAChE,MAAM,mBAAmB,GAAG,cAAc,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,EAAE,CAAC,CAAC,CAAC;IACtE,MAAM,WAAW,GAAG,gBAAgB,GAAG,mBAAmB,CAAC;IAC3D,MAAM,KAAK,GAAG,gBAAgB,CAAC,OAAO,CAAC,CAAC;IACxC,MAAM,MAAM,GAAG,eAAe,CAAC,OAAO,CAAC,CAAC;IACxC,MAAM,QAAQ,GAAG,OAAO,CAAC,iBAAiB,IAAI,2BAA2B,CAAC;IAC1E,4EAA4E;IAC5E,wEAAwE;IACxE,4CAA4C;IAC5C,MAAM,mBAAmB,GAAG,KAAK,GAAG,MAAM,GAAG,QAAQ,CAAC;IAEtD,OAAO;QACL,SAAS,EAAE,OAAO,CAAC,SAAS;QAC5B,KAAK,EAAE,OAAO,CAAC,KAAK;QACpB,KAAK,EAAE,OAAO,CAAC,KAAK,CAAC,MAAM;QAC3B,gBAAgB;QAChB,mBAAmB;QACnB,WAAW;QACX,aAAa,EAAE,gBAAgB,KAAK,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,WAAW,GAAG,gBAAgB;QAC7E,gBAAgB,EAAE,KAAK;QACvB,eAAe,EAAE,MAAM;QACvB,mBAAmB;QACnB,sBAAsB,EACpB,mBAAmB,KAAK,CAAC;YACvB,CAAC,CAAC,IAAI;YACN,CAAC,CAAC,CAAC,mBAAmB,GAAG,mBAAmB,CAAC,GAAG,mBAAmB;QACvE,WAAW;QACX,cAAc;QACd,wEAAwE;QACxE,yDAAyD;QACzD,SAAS,EAAE,WAAW,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,WAAW,CAAC,WAAW,CAAC,MAAM,GAAG,CAAC,CAAE,GAAG,WAAW,CAAC,CAAC,CAAE;QAC9F,YAAY,EAAE,cAAc,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,cAAc,CAAC,cAAc,CAAC,MAAM,GAAG,CAAC,CAAE,GAAG,cAAc,CAAC,CAAC,CAAE;QAC7G,8DAA8D;QAC9D,MAAM,EAAE,UAAU,CAAC,WAAW,EAAE,cAAc,CAAC;QAC/C,aAAa,EAAE;YACb,IAAI,EAAE,QAAQ,CAAC,WAAW,CAAC;YAC3B,OAAO,EAAE,QAAQ,CAAC,cAAc,CAAC;SAClC;KACF,CAAC;AACJ,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,oBAAoB,CAAC,MAAyB;IAI5D,MAAM,OAAO,GAAa,EAAE,CAAC;IAC7B,IAAI,MAAM,CAAC,KAAK,GAAG,CAAC,EAAE,CAAC;QACrB,OAAO,CAAC,IAAI,CAAC,mDAAmD,CAAC,CAAC;IACpE,CAAC;IACD,IAAI,MAAM,CAAC,SAAS,IAAI,CAAC,EAAE,CAAC;QAC1B,OAAO,CAAC,IAAI,CACV,sFAAsF;YACpF,wEAAwE,CAC3E,CAAC;IACJ,CAAC;IACD,IAAI,MAAM,CAAC,MAAM,CAAC,MAAM,KAAK,IAAI,IAAI,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,MAAM,CAAC,MAAM,CAAC,GAAG,kBAAkB,EAAE,CAAC;QACzF,OAAO,CAAC,IAAI,CACV,cAAc,MAAM,CAAC,MAAM,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC,CAAC,oBAAoB,kBAAkB,6BAA6B,CACjH,CAAC;IACJ,CAAC;IACD,OAAO,EAAE,MAAM,EAAE,OAAO,CAAC,MAAM,KAAK,CAAC,EAAE,OAAO,EAAE,CAAC;AACnD,CAAC"}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Rendering a verdict for a human reader.
|
|
3
|
+
*
|
|
4
|
+
* ── The formatting rule that matters ─────────────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* A percentage is printed on exactly two paths: a `saving` and a
|
|
7
|
+
* `regression`. Every refusal prints the gates instead, because a reader
|
|
8
|
+
* shown "39% saved" and a reader shown "INERT: the state file was never
|
|
9
|
+
* written" take completely different actions, and only one of them is
|
|
10
|
+
* correct.
|
|
11
|
+
*
|
|
12
|
+
* @non-paper — measurement infrastructure, not the paper's evaluation.
|
|
13
|
+
*/
|
|
14
|
+
import type { ArmId, ArmRecord } from './record.js';
|
|
15
|
+
import type { VerdictResult } from './verdict.js';
|
|
16
|
+
/**
|
|
17
|
+
* A per-arm table: n, median, spread, and engagement.
|
|
18
|
+
*
|
|
19
|
+
* The engagement column is not decoration. It is the fastest way for a reader
|
|
20
|
+
* to see that an arm measured nothing, and it is the column whose absence
|
|
21
|
+
* produced the original false positive.
|
|
22
|
+
*/
|
|
23
|
+
export declare function formatArmTable(arms: ReadonlyMap<ArmId, ArmRecord>): string;
|
|
24
|
+
/** The full human-readable report for one experiment. */
|
|
25
|
+
export declare function formatVerdict(result: VerdictResult, table: string): string;
|
|
26
|
+
//# sourceMappingURL=report.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"report.d.ts","sourceRoot":"","sources":["../../src/ab/report.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AAGH,OAAO,KAAK,EAAE,KAAK,EAAE,SAAS,EAAE,MAAM,aAAa,CAAC;AAGpD,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,cAAc,CAAC;AAgBlD;;;;;;GAMG;AACH,wBAAgB,cAAc,CAAC,IAAI,EAAE,WAAW,CAAC,KAAK,EAAE,SAAS,CAAC,GAAG,MAAM,CAwB1E;AAiBD,yDAAyD;AACzD,wBAAgB,aAAa,CAC3B,MAAM,EAAE,aAAa,EACrB,KAAK,EAAE,MAAM,GACZ,MAAM,CA0BR"}
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Rendering a verdict for a human reader.
|
|
3
|
+
*
|
|
4
|
+
* ── The formatting rule that matters ─────────────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* A percentage is printed on exactly two paths: a `saving` and a
|
|
7
|
+
* `regression`. Every refusal prints the gates instead, because a reader
|
|
8
|
+
* shown "39% saved" and a reader shown "INERT: the state file was never
|
|
9
|
+
* written" take completely different actions, and only one of them is
|
|
10
|
+
* correct.
|
|
11
|
+
*
|
|
12
|
+
* @non-paper — measurement infrastructure, not the paper's evaluation.
|
|
13
|
+
*/
|
|
14
|
+
import { promptTokens } from './record.js';
|
|
15
|
+
import { describe } from './stats-core.js';
|
|
16
|
+
const HEADLINE = {
|
|
17
|
+
saving: 'SAVING',
|
|
18
|
+
regression: 'REGRESSION',
|
|
19
|
+
'no-effect': 'NO-EFFECT',
|
|
20
|
+
inert: 'INERT',
|
|
21
|
+
'not-comparable': 'NOT-COMPARABLE',
|
|
22
|
+
invalid: 'INVALID',
|
|
23
|
+
};
|
|
24
|
+
/** Left-pad a label to a fixed width so the table stays aligned. */
|
|
25
|
+
function cell(text, width) {
|
|
26
|
+
return text.length >= width ? text : text + ' '.repeat(width - text.length);
|
|
27
|
+
}
|
|
28
|
+
/**
|
|
29
|
+
* A per-arm table: n, median, spread, and engagement.
|
|
30
|
+
*
|
|
31
|
+
* The engagement column is not decoration. It is the fastest way for a reader
|
|
32
|
+
* to see that an arm measured nothing, and it is the column whose absence
|
|
33
|
+
* produced the original false positive.
|
|
34
|
+
*/
|
|
35
|
+
export function formatArmTable(arms) {
|
|
36
|
+
const rows = [
|
|
37
|
+
['arm', 'n', 'prompt median', 'MAD', 'spread', 'state writes', 'turns']
|
|
38
|
+
.map((h) => cell(h, 13))
|
|
39
|
+
.join(' '),
|
|
40
|
+
];
|
|
41
|
+
for (const [arm, record] of arms) {
|
|
42
|
+
const promptTokensSeen = record.runs.map((run) => promptTokens(run.usage));
|
|
43
|
+
const distribution = armDistribution(promptTokensSeen);
|
|
44
|
+
const writes = record.runs.reduce((n, run) => n + run.engagement.writes, 0);
|
|
45
|
+
const turns = record.runs.reduce((n, run) => n + run.work.turns, 0);
|
|
46
|
+
rows.push([
|
|
47
|
+
cell(arm, 13),
|
|
48
|
+
cell(String(distribution.n), 13),
|
|
49
|
+
cell(String(distribution.median), 13),
|
|
50
|
+
cell(String(distribution.mad), 13),
|
|
51
|
+
cell(`±${(distribution.relativeMad * 100).toFixed(0)}%`, 13),
|
|
52
|
+
cell(String(writes), 13),
|
|
53
|
+
cell(String(turns), 13),
|
|
54
|
+
].join(' '));
|
|
55
|
+
}
|
|
56
|
+
return rows.join('\n');
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* Distribution of one arm, tolerating an arm with no runs.
|
|
60
|
+
*
|
|
61
|
+
* `describe` throws on an empty sample, which is the right behaviour for a
|
|
62
|
+
* gate that must not compare an unrun arm. A table is a different contract:
|
|
63
|
+
* a diagnostic that crashes on the arm it is meant to diagnose is useless,
|
|
64
|
+
* so an absent arm renders as zeros here. The gate still refuses the run.
|
|
65
|
+
*/
|
|
66
|
+
function armDistribution(values) {
|
|
67
|
+
if (values.length === 0) {
|
|
68
|
+
return { n: 0, min: 0, median: 0, max: 0, mad: 0, relativeMad: 0, sorted: [] };
|
|
69
|
+
}
|
|
70
|
+
return describe(values);
|
|
71
|
+
}
|
|
72
|
+
/** The full human-readable report for one experiment. */
|
|
73
|
+
export function formatVerdict(result, table) {
|
|
74
|
+
const lines = [table, ''];
|
|
75
|
+
const measured = result.verdict === 'saving' || result.verdict === 'regression';
|
|
76
|
+
if (measured && result.effect !== null) {
|
|
77
|
+
const relative = result.effect.relative === null
|
|
78
|
+
? 'n/a (control median is 0)'
|
|
79
|
+
: `${(result.effect.relative * 100).toFixed(1)}%`;
|
|
80
|
+
lines.push(`${HEADLINE[result.verdict]}: ${relative} of prompt tokens ` +
|
|
81
|
+
`(${result.effect.difference > 0 ? '' : '+'}${result.effect.difference} tokens at the median)` +
|
|
82
|
+
(result.effect.inMads === null
|
|
83
|
+
? ', exact (both arms had zero spread)'
|
|
84
|
+
: `, ${Math.abs(result.effect.inMads).toFixed(1)} MADs`));
|
|
85
|
+
}
|
|
86
|
+
else {
|
|
87
|
+
lines.push(`${HEADLINE[result.verdict]}: no effect size reported.`);
|
|
88
|
+
}
|
|
89
|
+
if (result.failures.length > 0) {
|
|
90
|
+
lines.push('');
|
|
91
|
+
lines.push('Gates that fired:');
|
|
92
|
+
for (const failure of result.failures) {
|
|
93
|
+
lines.push(` - [${failure.gate}] ${failure.detail}`);
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
return lines.join('\n');
|
|
97
|
+
}
|
|
98
|
+
//# sourceMappingURL=report.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"report.js","sourceRoot":"","sources":["../../src/ab/report.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;GAYG;AAEH,OAAO,EAAE,YAAY,EAAE,MAAM,aAAa,CAAC;AAE3C,OAAO,EAAE,QAAQ,EAAE,MAAM,iBAAiB,CAAC;AAI3C,MAAM,QAAQ,GAAuD;IACnE,MAAM,EAAE,QAAQ;IAChB,UAAU,EAAE,YAAY;IACxB,WAAW,EAAE,WAAW;IACxB,KAAK,EAAE,OAAO;IACd,gBAAgB,EAAE,gBAAgB;IAClC,OAAO,EAAE,SAAS;CACnB,CAAC;AAEF,oEAAoE;AACpE,SAAS,IAAI,CAAC,IAAY,EAAE,KAAa;IACvC,OAAO,IAAI,CAAC,MAAM,IAAI,KAAK,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,IAAI,GAAG,GAAG,CAAC,MAAM,CAAC,KAAK,GAAG,IAAI,CAAC,MAAM,CAAC,CAAC;AAC9E,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,cAAc,CAAC,IAAmC;IAChE,MAAM,IAAI,GAAa;QACrB,CAAC,KAAK,EAAE,GAAG,EAAE,eAAe,EAAE,KAAK,EAAE,QAAQ,EAAE,cAAc,EAAE,OAAO,CAAC;aACpE,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,IAAI,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC;aACvB,IAAI,CAAC,GAAG,CAAC;KACb,CAAC;IACF,KAAK,MAAM,CAAC,GAAG,EAAE,MAAM,CAAC,IAAI,IAAI,EAAE,CAAC;QACjC,MAAM,gBAAgB,GAAG,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,YAAY,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,CAAC;QAC3E,MAAM,YAAY,GAAG,eAAe,CAAC,gBAAgB,CAAC,CAAC;QACvD,MAAM,MAAM,GAAG,MAAM,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,GAAG,EAAE,EAAE,CAAC,CAAC,GAAG,GAAG,CAAC,UAAU,CAAC,MAAM,EAAE,CAAC,CAAC,CAAC;QAC5E,MAAM,KAAK,GAAG,MAAM,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,GAAG,EAAE,EAAE,CAAC,CAAC,GAAG,GAAG,CAAC,IAAI,CAAC,KAAK,EAAE,CAAC,CAAC,CAAC;QACpE,IAAI,CAAC,IAAI,CACP;YACE,IAAI,CAAC,GAAG,EAAE,EAAE,CAAC;YACb,IAAI,CAAC,MAAM,CAAC,YAAY,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC;YAChC,IAAI,CAAC,MAAM,CAAC,YAAY,CAAC,MAAM,CAAC,EAAE,EAAE,CAAC;YACrC,IAAI,CAAC,MAAM,CAAC,YAAY,CAAC,GAAG,CAAC,EAAE,EAAE,CAAC;YAClC,IAAI,CAAC,IAAI,CAAC,YAAY,CAAC,WAAW,GAAG,GAAG,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC;YAC5D,IAAI,CAAC,MAAM,CAAC,MAAM,CAAC,EAAE,EAAE,CAAC;YACxB,IAAI,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,EAAE,CAAC;SACxB,CAAC,IAAI,CAAC,GAAG,CAAC,CACZ,CAAC;IACJ,CAAC;IACD,OAAO,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AACzB,CAAC;AAED;;;;;;;GAOG;AACH,SAAS,eAAe,CAAC,MAAyB;IAChD,IAAI,MAAM,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACxB,OAAO,EAAE,CAAC,EAAE,CAAC,EAAE,GAAG,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,GAAG,EAAE,CAAC,EAAE,GAAG,EAAE,CAAC,EAAE,WAAW,EAAE,CAAC,EAAE,MAAM,EAAE,EAAE,EAAE,CAAC;IACjF,CAAC;IACD,OAAO,QAAQ,CAAC,MAAM,CAAC,CAAC;AAC1B,CAAC;AAED,yDAAyD;AACzD,MAAM,UAAU,aAAa,CAC3B,MAAqB,EACrB,KAAa;IAEb,MAAM,KAAK,GAAa,CAAC,KAAK,EAAE,EAAE,CAAC,CAAC;IACpC,MAAM,QAAQ,GAAG,MAAM,CAAC,OAAO,KAAK,QAAQ,IAAI,MAAM,CAAC,OAAO,KAAK,YAAY,CAAC;IAChF,IAAI,QAAQ,IAAI,MAAM,CAAC,MAAM,KAAK,IAAI,EAAE,CAAC;QACvC,MAAM,QAAQ,GACZ,MAAM,CAAC,MAAM,CAAC,QAAQ,KAAK,IAAI;YAC7B,CAAC,CAAC,2BAA2B;YAC7B,CAAC,CAAC,GAAG,CAAC,MAAM,CAAC,MAAM,CAAC,QAAQ,GAAG,GAAG,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,GAAG,CAAC;QACtD,KAAK,CAAC,IAAI,CACR,GAAG,QAAQ,CAAC,MAAM,CAAC,OAAO,CAAC,KAAK,QAAQ,oBAAoB;YAC1D,IAAI,MAAM,CAAC,MAAM,CAAC,UAAU,GAAG,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,GAAG,MAAM,CAAC,MAAM,CAAC,UAAU,wBAAwB;YAC9F,CAAC,MAAM,CAAC,MAAM,CAAC,MAAM,KAAK,IAAI;gBAC5B,CAAC,CAAC,qCAAqC;gBACvC,CAAC,CAAC,KAAK,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAC7D,CAAC;IACJ,CAAC;SAAM,CAAC;QACN,KAAK,CAAC,IAAI,CAAC,GAAG,QAAQ,CAAC,MAAM,CAAC,OAAO,CAAC,4BAA4B,CAAC,CAAC;IACtE,CAAC;IACD,IAAI,MAAM,CAAC,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QAC/B,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;QACf,KAAK,CAAC,IAAI,CAAC,mBAAmB,CAAC,CAAC;QAChC,KAAK,MAAM,OAAO,IAAI,MAAM,CAAC,QAAQ,EAAE,CAAC;YACtC,KAAK,CAAC,IAAI,CAAC,QAAQ,OAAO,CAAC,IAAI,KAAK,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;QACxD,CAAC;IACH,CAAC;IACD,OAAO,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAC1B,CAAC"}
|