@skillstate/bench 3.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -2
- package/dist/ab/engagement.d.ts +70 -0
- package/dist/ab/engagement.d.ts.map +1 -0
- package/dist/ab/engagement.js +102 -0
- package/dist/ab/engagement.js.map +1 -0
- package/dist/ab/index.d.ts +28 -0
- package/dist/ab/index.d.ts.map +1 -0
- package/dist/ab/index.js +20 -0
- package/dist/ab/index.js.map +1 -0
- package/dist/ab/opencode-usage.d.ts +52 -0
- package/dist/ab/opencode-usage.d.ts.map +1 -0
- package/dist/ab/opencode-usage.js +85 -0
- package/dist/ab/opencode-usage.js.map +1 -0
- package/dist/ab/record.d.ts +173 -0
- package/dist/ab/record.d.ts.map +1 -0
- package/dist/ab/record.js +98 -0
- package/dist/ab/record.js.map +1 -0
- package/dist/ab/replay.d.ts +193 -0
- package/dist/ab/replay.d.ts.map +1 -0
- package/dist/ab/replay.js +149 -0
- package/dist/ab/replay.js.map +1 -0
- package/dist/ab/report.d.ts +26 -0
- package/dist/ab/report.d.ts.map +1 -0
- package/dist/ab/report.js +98 -0
- package/dist/ab/report.js.map +1 -0
- package/dist/ab/stats-core.d.ts +84 -0
- package/dist/ab/stats-core.d.ts.map +1 -0
- package/dist/ab/stats-core.js +93 -0
- package/dist/ab/stats-core.js.map +1 -0
- package/dist/ab/survey.d.ts +131 -0
- package/dist/ab/survey.d.ts.map +1 -0
- package/dist/ab/survey.js +143 -0
- package/dist/ab/survey.js.map +1 -0
- package/dist/ab/usage.d.ts +99 -0
- package/dist/ab/usage.d.ts.map +1 -0
- package/dist/ab/usage.js +98 -0
- package/dist/ab/usage.js.map +1 -0
- package/dist/ab/verdict.d.ts +88 -0
- package/dist/ab/verdict.d.ts.map +1 -0
- package/dist/ab/verdict.js +251 -0
- package/dist/ab/verdict.js.map +1 -0
- package/dist/ab-cli.d.ts +35 -0
- package/dist/ab-cli.d.ts.map +1 -0
- package/dist/ab-cli.js +161 -0
- package/dist/ab-cli.js.map +1 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/package.json +5 -1
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Distribution summaries for a set of runs.
|
|
3
|
+
*
|
|
4
|
+
* ── Why the median and not the mean ──────────────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* Run-to-run variance in a non-deterministic host is large — the old A/B
|
|
7
|
+
* measured the same arm twice and got 42 364 and 71 590 input tokens, a 69%
|
|
8
|
+
* swing between a mid-run reading and the final one. A mean over a handful
|
|
9
|
+
* of such samples is dominated by whichever sample was luckiest, and a
|
|
10
|
+
* percentage computed from it inherits that luck.
|
|
11
|
+
*
|
|
12
|
+
* The median is reported instead, together with the median absolute
|
|
13
|
+
* deviation (MAD), so a reader sees the SPREAD next to the effect. An effect
|
|
14
|
+
* smaller than the spread is not an effect, and the harness in `verdict.ts`
|
|
15
|
+
* refuses to call it one.
|
|
16
|
+
*
|
|
17
|
+
* These are deliberately dependency-free and deterministic: same input, same
|
|
18
|
+
* output, no RNG anywhere. That is what lets the gates be tested against the
|
|
19
|
+
* real historical numbers.
|
|
20
|
+
*
|
|
21
|
+
* @non-paper — measurement infrastructure, not the paper's evaluation.
|
|
22
|
+
*/
|
|
23
|
+
/**
|
|
24
|
+
* Median of a non-empty numeric sample.
|
|
25
|
+
*
|
|
26
|
+
* Requires a non-empty array and throws on an empty one: an empty sample has
|
|
27
|
+
* no median, and inventing one (0, or NaN) would let a silently-unrun arm
|
|
28
|
+
* compare as if it had run. The gates validate sample size before calling.
|
|
29
|
+
*/
|
|
30
|
+
export declare function median(values: readonly number[]): number;
|
|
31
|
+
/**
|
|
32
|
+
* Median absolute deviation — a spread measure that ignores outliers.
|
|
33
|
+
*
|
|
34
|
+
* Subtracting the median first is what makes it robust: one runaway run
|
|
35
|
+
* inflates a standard deviation enough to hide a real effect, but barely
|
|
36
|
+
* moves the MAD, because the runaway run's own deviation is measured against
|
|
37
|
+
* a median that the median itself did not follow.
|
|
38
|
+
*/
|
|
39
|
+
export declare function mad(values: readonly number[]): number;
|
|
40
|
+
/** The spread of a sample, as a fraction of its median. Zero when the median is 0. */
|
|
41
|
+
export declare function relativeMad(values: readonly number[]): number;
|
|
42
|
+
/** A sample summarised for reporting. */
|
|
43
|
+
export interface Distribution {
|
|
44
|
+
readonly n: number;
|
|
45
|
+
readonly min: number;
|
|
46
|
+
readonly median: number;
|
|
47
|
+
readonly max: number;
|
|
48
|
+
/** Median absolute deviation, in the same units as the sample. */
|
|
49
|
+
readonly mad: number;
|
|
50
|
+
/** MAD as a fraction of the median. 0 when the median is 0. */
|
|
51
|
+
readonly relativeMad: number;
|
|
52
|
+
/** The raw values, sorted ascending, so a reader can check the summary. */
|
|
53
|
+
readonly sorted: readonly number[];
|
|
54
|
+
}
|
|
55
|
+
/** Summarise a sample. Throws on an empty one — see {@link median}. */
|
|
56
|
+
export declare function describe(values: readonly number[]): Distribution;
|
|
57
|
+
/**
|
|
58
|
+
* A relative effect size, robust to scale and to spread.
|
|
59
|
+
*
|
|
60
|
+
* `effect = (control − instrumented) / pooledSpread`, where `pooledSpread` is
|
|
61
|
+
* the larger of the two arms' MADs. The sign carries the direction (positive
|
|
62
|
+
* means the instrumented arm spent fewer prompt tokens); the magnitude is the
|
|
63
|
+
* number of MADs the gap is worth.
|
|
64
|
+
*
|
|
65
|
+
* When both arms have zero spread the comparison is exact and the ratio is
|
|
66
|
+
* undefined, so `exact: true` is returned with the plain difference. Callers
|
|
67
|
+
* must branch on `exact` rather than dividing by the missing ratio — that
|
|
68
|
+
* branch is where a harness quietly invents an effect.
|
|
69
|
+
*/
|
|
70
|
+
export interface EffectSize {
|
|
71
|
+
/** False when either arm had zero spread and the ratio cannot be formed. */
|
|
72
|
+
readonly exact: boolean;
|
|
73
|
+
/** `control − instrumented`, in tokens. Positive favours the instrumented arm. */
|
|
74
|
+
readonly difference: number;
|
|
75
|
+
/** Difference as a fraction of the control median. Null when the median is 0. */
|
|
76
|
+
readonly relative: number | null;
|
|
77
|
+
/** Difference in units of pooled MAD. Null when `exact` is false. */
|
|
78
|
+
readonly inMads: number | null;
|
|
79
|
+
}
|
|
80
|
+
/** Minimum effect, in MADs, the harness is willing to call a signal. */
|
|
81
|
+
export declare const MIN_EFFECT_IN_MADS = 1;
|
|
82
|
+
/** Compare two arms' prompt-token distributions. Throws if either sample is empty. */
|
|
83
|
+
export declare function effectSize(control: readonly number[], instrumented: readonly number[]): EffectSize;
|
|
84
|
+
//# sourceMappingURL=stats-core.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"stats-core.d.ts","sourceRoot":"","sources":["../../src/ab/stats-core.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH;;;;;;GAMG;AACH,wBAAgB,MAAM,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,CASxD;AAED;;;;;;;GAOG;AACH,wBAAgB,GAAG,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,CAMrD;AAED,sFAAsF;AACtF,wBAAgB,WAAW,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,CAG7D;AAED,yCAAyC;AACzC,MAAM,WAAW,YAAY;IAC3B,QAAQ,CAAC,CAAC,EAAE,MAAM,CAAC;IACnB,QAAQ,CAAC,GAAG,EAAE,MAAM,CAAC;IACrB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,QAAQ,CAAC,GAAG,EAAE,MAAM,CAAC;IACrB,kEAAkE;IAClE,QAAQ,CAAC,GAAG,EAAE,MAAM,CAAC;IACrB,+DAA+D;IAC/D,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAC7B,2EAA2E;IAC3E,QAAQ,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,CAAC;CACpC;AAED,uEAAuE;AACvE,wBAAgB,QAAQ,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,GAAG,YAAY,CAchE;AAED;;;;;;;;;;;;GAYG;AACH,MAAM,WAAW,UAAU;IACzB,4EAA4E;IAC5E,QAAQ,CAAC,KAAK,EAAE,OAAO,CAAC;IACxB,kFAAkF;IAClF,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B,iFAAiF;IACjF,QAAQ,CAAC,QAAQ,EAAE,MAAM,GAAG,IAAI,CAAC;IACjC,qEAAqE;IACrE,QAAQ,CAAC,MAAM,EAAE,MAAM,GAAG,IAAI,CAAC;CAChC;AAED,wEAAwE;AACxE,eAAO,MAAM,kBAAkB,IAAI,CAAC;AAEpC,sFAAsF;AACtF,wBAAgB,UAAU,CACxB,OAAO,EAAE,SAAS,MAAM,EAAE,EAC1B,YAAY,EAAE,SAAS,MAAM,EAAE,GAC9B,UAAU,CAcZ"}
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Distribution summaries for a set of runs.
|
|
3
|
+
*
|
|
4
|
+
* ── Why the median and not the mean ──────────────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* Run-to-run variance in a non-deterministic host is large — the old A/B
|
|
7
|
+
* measured the same arm twice and got 42 364 and 71 590 input tokens, a 69%
|
|
8
|
+
* swing between a mid-run reading and the final one. A mean over a handful
|
|
9
|
+
* of such samples is dominated by whichever sample was luckiest, and a
|
|
10
|
+
* percentage computed from it inherits that luck.
|
|
11
|
+
*
|
|
12
|
+
* The median is reported instead, together with the median absolute
|
|
13
|
+
* deviation (MAD), so a reader sees the SPREAD next to the effect. An effect
|
|
14
|
+
* smaller than the spread is not an effect, and the harness in `verdict.ts`
|
|
15
|
+
* refuses to call it one.
|
|
16
|
+
*
|
|
17
|
+
* These are deliberately dependency-free and deterministic: same input, same
|
|
18
|
+
* output, no RNG anywhere. That is what lets the gates be tested against the
|
|
19
|
+
* real historical numbers.
|
|
20
|
+
*
|
|
21
|
+
* @non-paper — measurement infrastructure, not the paper's evaluation.
|
|
22
|
+
*/
|
|
23
|
+
/**
|
|
24
|
+
* Median of a non-empty numeric sample.
|
|
25
|
+
*
|
|
26
|
+
* Requires a non-empty array and throws on an empty one: an empty sample has
|
|
27
|
+
* no median, and inventing one (0, or NaN) would let a silently-unrun arm
|
|
28
|
+
* compare as if it had run. The gates validate sample size before calling.
|
|
29
|
+
*/
|
|
30
|
+
export function median(values) {
|
|
31
|
+
if (values.length === 0) {
|
|
32
|
+
throw new RangeError('median of an empty sample is undefined');
|
|
33
|
+
}
|
|
34
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
35
|
+
const middle = sorted.length >> 1;
|
|
36
|
+
return sorted.length % 2 === 1
|
|
37
|
+
? sorted[middle]
|
|
38
|
+
: (sorted[middle - 1] + sorted[middle]) / 2;
|
|
39
|
+
}
|
|
40
|
+
/**
|
|
41
|
+
* Median absolute deviation — a spread measure that ignores outliers.
|
|
42
|
+
*
|
|
43
|
+
* Subtracting the median first is what makes it robust: one runaway run
|
|
44
|
+
* inflates a standard deviation enough to hide a real effect, but barely
|
|
45
|
+
* moves the MAD, because the runaway run's own deviation is measured against
|
|
46
|
+
* a median that the median itself did not follow.
|
|
47
|
+
*/
|
|
48
|
+
export function mad(values) {
|
|
49
|
+
if (values.length === 0) {
|
|
50
|
+
throw new RangeError('MAD of an empty sample is undefined');
|
|
51
|
+
}
|
|
52
|
+
const center = median(values);
|
|
53
|
+
return median(values.map((value) => Math.abs(value - center)));
|
|
54
|
+
}
|
|
55
|
+
/** The spread of a sample, as a fraction of its median. Zero when the median is 0. */
|
|
56
|
+
export function relativeMad(values) {
|
|
57
|
+
const center = median(values);
|
|
58
|
+
return center === 0 ? 0 : mad(values) / center;
|
|
59
|
+
}
|
|
60
|
+
/** Summarise a sample. Throws on an empty one — see {@link median}. */
|
|
61
|
+
export function describe(values) {
|
|
62
|
+
if (values.length === 0) {
|
|
63
|
+
throw new RangeError('cannot describe an empty sample');
|
|
64
|
+
}
|
|
65
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
66
|
+
return {
|
|
67
|
+
n: values.length,
|
|
68
|
+
min: sorted[0],
|
|
69
|
+
median: median(values),
|
|
70
|
+
max: sorted[sorted.length - 1],
|
|
71
|
+
mad: mad(values),
|
|
72
|
+
relativeMad: relativeMad(values),
|
|
73
|
+
sorted,
|
|
74
|
+
};
|
|
75
|
+
}
|
|
76
|
+
/** Minimum effect, in MADs, the harness is willing to call a signal. */
|
|
77
|
+
export const MIN_EFFECT_IN_MADS = 1;
|
|
78
|
+
/** Compare two arms' prompt-token distributions. Throws if either sample is empty. */
|
|
79
|
+
export function effectSize(control, instrumented) {
|
|
80
|
+
if (control.length === 0 || instrumented.length === 0) {
|
|
81
|
+
throw new RangeError('effect size needs a non-empty sample in both arms');
|
|
82
|
+
}
|
|
83
|
+
const controlMedian = median(control);
|
|
84
|
+
const instrumentedMedian = median(instrumented);
|
|
85
|
+
const difference = controlMedian - instrumentedMedian;
|
|
86
|
+
const relative = controlMedian === 0 ? null : difference / controlMedian;
|
|
87
|
+
const spread = Math.max(mad(control), mad(instrumented));
|
|
88
|
+
if (spread === 0) {
|
|
89
|
+
return { exact: true, difference, relative, inMads: null };
|
|
90
|
+
}
|
|
91
|
+
return { exact: false, difference, relative, inMads: difference / spread };
|
|
92
|
+
}
|
|
93
|
+
//# sourceMappingURL=stats-core.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"stats-core.js","sourceRoot":"","sources":["../../src/ab/stats-core.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH;;;;;;GAMG;AACH,MAAM,UAAU,MAAM,CAAC,MAAyB;IAC9C,IAAI,MAAM,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACxB,MAAM,IAAI,UAAU,CAAC,wCAAwC,CAAC,CAAC;IACjE,CAAC;IACD,MAAM,MAAM,GAAG,CAAC,GAAG,MAAM,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IACjD,MAAM,MAAM,GAAG,MAAM,CAAC,MAAM,IAAI,CAAC,CAAC;IAClC,OAAO,MAAM,CAAC,MAAM,GAAG,CAAC,KAAK,CAAC;QAC5B,CAAC,CAAC,MAAM,CAAC,MAAM,CAAE;QACjB,CAAC,CAAC,CAAC,MAAM,CAAC,MAAM,GAAG,CAAC,CAAE,GAAG,MAAM,CAAC,MAAM,CAAE,CAAC,GAAG,CAAC,CAAC;AAClD,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,GAAG,CAAC,MAAyB;IAC3C,IAAI,MAAM,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACxB,MAAM,IAAI,UAAU,CAAC,qCAAqC,CAAC,CAAC;IAC9D,CAAC;IACD,MAAM,MAAM,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC;IAC9B,OAAO,MAAM,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,GAAG,MAAM,CAAC,CAAC,CAAC,CAAC;AACjE,CAAC;AAED,sFAAsF;AACtF,MAAM,UAAU,WAAW,CAAC,MAAyB;IACnD,MAAM,MAAM,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC;IAC9B,OAAO,MAAM,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,MAAM,CAAC,GAAG,MAAM,CAAC;AACjD,CAAC;AAgBD,uEAAuE;AACvE,MAAM,UAAU,QAAQ,CAAC,MAAyB;IAChD,IAAI,MAAM,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACxB,MAAM,IAAI,UAAU,CAAC,iCAAiC,CAAC,CAAC;IAC1D,CAAC;IACD,MAAM,MAAM,GAAG,CAAC,GAAG,MAAM,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IACjD,OAAO;QACL,CAAC,EAAE,MAAM,CAAC,MAAM;QAChB,GAAG,EAAE,MAAM,CAAC,CAAC,CAAE;QACf,MAAM,EAAE,MAAM,CAAC,MAAM,CAAC;QACtB,GAAG,EAAE,MAAM,CAAC,MAAM,CAAC,MAAM,GAAG,CAAC,CAAE;QAC/B,GAAG,EAAE,GAAG,CAAC,MAAM,CAAC;QAChB,WAAW,EAAE,WAAW,CAAC,MAAM,CAAC;QAChC,MAAM;KACP,CAAC;AACJ,CAAC;AA0BD,wEAAwE;AACxE,MAAM,CAAC,MAAM,kBAAkB,GAAG,CAAC,CAAC;AAEpC,sFAAsF;AACtF,MAAM,UAAU,UAAU,CACxB,OAA0B,EAC1B,YAA+B;IAE/B,IAAI,OAAO,CAAC,MAAM,KAAK,CAAC,IAAI,YAAY,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACtD,MAAM,IAAI,UAAU,CAAC,mDAAmD,CAAC,CAAC;IAC5E,CAAC;IACD,MAAM,aAAa,GAAG,MAAM,CAAC,OAAO,CAAC,CAAC;IACtC,MAAM,kBAAkB,GAAG,MAAM,CAAC,YAAY,CAAC,CAAC;IAChD,MAAM,UAAU,GAAG,aAAa,GAAG,kBAAkB,CAAC;IACtD,MAAM,QAAQ,GAAG,aAAa,KAAK,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,UAAU,GAAG,aAAa,CAAC;IACzE,MAAM,MAAM,GAAG,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,OAAO,CAAC,EAAE,GAAG,CAAC,YAAY,CAAC,CAAC,CAAC;IAEzD,IAAI,MAAM,KAAK,CAAC,EAAE,CAAC;QACjB,OAAO,EAAE,KAAK,EAAE,IAAI,EAAE,UAAU,EAAE,QAAQ,EAAE,MAAM,EAAE,IAAI,EAAE,CAAC;IAC7D,CAAC;IACD,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,UAAU,EAAE,QAAQ,EAAE,MAAM,EAAE,UAAU,GAAG,MAAM,EAAE,CAAC;AAC7E,CAAC"}
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A survey over a real host's own session store.
|
|
3
|
+
*
|
|
4
|
+
* ── What this measures, and what it does not ──────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* It answers ONE question, on real data, with no model in the loop:
|
|
7
|
+
*
|
|
8
|
+
* *given a real session's real token accounting, what would a bounded
|
|
9
|
+
* Aₜ = (P, Σₜ, Oₜ) prompt have cost on those same steps?*
|
|
10
|
+
*
|
|
11
|
+
* It does NOT answer whether an agent given that bounded prompt still does the
|
|
12
|
+
* work. That is an outcome question, it needs a live model, and no amount of
|
|
13
|
+
* offline arithmetic substitutes for it. A survey that reported a saving as if
|
|
14
|
+
* it established the claim would be repeating the original error with extra
|
|
15
|
+
* steps — the original A/B reported a number for a run whose integration had
|
|
16
|
+
* never engaged, and a cost-only win with no task completion is worth nothing.
|
|
17
|
+
*
|
|
18
|
+
* The two halves are complementary: this establishes that the cost side is
|
|
19
|
+
* real and large, the A/B establishes that the work still gets done.
|
|
20
|
+
*
|
|
21
|
+
* ── Why the host's own store ─────────────────────────────────────────────
|
|
22
|
+
*
|
|
23
|
+
* OpenCode records per-message `tokens` — `input`, `cache.read`, `output`,
|
|
24
|
+
* `reasoning`. Reading them means the numbers are the host's accounting for a
|
|
25
|
+
* real session rather than our reconstruction of it, and it works with no live
|
|
26
|
+
* model, which is what makes it usable when a provider quota is exhausted.
|
|
27
|
+
*
|
|
28
|
+
* `cache.read` is the field that carries the finding. A growing transcript is
|
|
29
|
+
* not merely expensive to send: the host caches the prefix, so re-reading
|
|
30
|
+
* history is billed as cache reads. Counting only fresh `input` would report a
|
|
31
|
+
* few percent and miss the entire effect.
|
|
32
|
+
*
|
|
33
|
+
* @non-paper — measurement infrastructure, not the paper's evaluation.
|
|
34
|
+
*/
|
|
35
|
+
import { assessReconstruction, reconstruct } from './replay.js';
|
|
36
|
+
import type { HostSession, ReconstructResult, StepUsage } from './replay.js';
|
|
37
|
+
/** The result of surveying a set of sessions. */
|
|
38
|
+
export interface Survey {
|
|
39
|
+
/** Sessions with enough steps to compare curves. */
|
|
40
|
+
readonly sessions: number;
|
|
41
|
+
/** Total steps across every session. */
|
|
42
|
+
readonly steps: number;
|
|
43
|
+
/** What the host actually spent on prompts, summed. */
|
|
44
|
+
readonly hostPromptTokens: number;
|
|
45
|
+
/** What bounded Aₜ prompts would have cost, summed. */
|
|
46
|
+
readonly boundedPromptTokens: number;
|
|
47
|
+
/** `host − bounded`, summed. */
|
|
48
|
+
readonly savedTokens: number;
|
|
49
|
+
/** `saved / host`, as a raw token count. Inflated; see {@link savedEffectiveFraction}. */
|
|
50
|
+
readonly savedFraction: number | null;
|
|
51
|
+
/** Fresh, uncached input tokens across the corpus. */
|
|
52
|
+
readonly freshInputTokens: number;
|
|
53
|
+
/** Cache-read tokens across the corpus. */
|
|
54
|
+
readonly cacheReadTokens: number;
|
|
55
|
+
/** Cache reads as a share of all prompt tokens. */
|
|
56
|
+
readonly cacheShare: number;
|
|
57
|
+
/**
|
|
58
|
+
* The corpus priced in input-equivalent tokens, cache reads discounted.
|
|
59
|
+
*
|
|
60
|
+
* The number to quote. The raw total counts a cache read as equal to a
|
|
61
|
+
* fresh input token, and on a cache-heavy corpus that inflates a saving by
|
|
62
|
+
* roughly an order of magnitude.
|
|
63
|
+
*/
|
|
64
|
+
readonly hostEffectiveTokens: number;
|
|
65
|
+
/**
|
|
66
|
+
* The saving priced properly. Always ≤ {@link savedFraction}, and the
|
|
67
|
+
* defensible one.
|
|
68
|
+
*/
|
|
69
|
+
readonly savedEffectiveFraction: number | null;
|
|
70
|
+
/** Per-session saving fractions, for a median. */
|
|
71
|
+
readonly perSession: readonly number[];
|
|
72
|
+
/** Median per-session saving. */
|
|
73
|
+
readonly medianSavedFraction: number;
|
|
74
|
+
/** Sessions whose prompt grew between first and last step. */
|
|
75
|
+
readonly grewCount: number;
|
|
76
|
+
/**
|
|
77
|
+
* Sessions where the bounded prompt would have cost MORE.
|
|
78
|
+
*
|
|
79
|
+
* The strongest number in the survey: a bounded prompt that never loses is
|
|
80
|
+
* a different claim from one that usually wins, and only the counter can
|
|
81
|
+
* distinguish them.
|
|
82
|
+
*
|
|
83
|
+
* CAVEAT when quoting it: this counts every session, including rows that
|
|
84
|
+
* recorded zero tokens — aborted runs and records written before
|
|
85
|
+
* accounting was populated. Such a session is not "a run that came out
|
|
86
|
+
* cheap", it is a run that never happened, and it will register as a loss
|
|
87
|
+
* for any bounded prompt. Filter on {@link spentSessions} before claiming
|
|
88
|
+
* "never loses", or the number is smaller and more honest.
|
|
89
|
+
*/
|
|
90
|
+
readonly boundedLosesCount: number;
|
|
91
|
+
/** Sessions where the host recorded any token spend at all. */
|
|
92
|
+
readonly spentSessions: number;
|
|
93
|
+
/** Per-session results, largest saving first. */
|
|
94
|
+
readonly results: readonly ReconstructResult[];
|
|
95
|
+
}
|
|
96
|
+
/** Options for {@link survey}. */
|
|
97
|
+
export interface SurveyOptions {
|
|
98
|
+
/** Tokens per bounded Aₜ prompt. Defaults to the paper's ~1.8k. */
|
|
99
|
+
readonly boundedPromptTokens?: number;
|
|
100
|
+
/** Include a session's full per-step arrays in `results`. Defaults to false. */
|
|
101
|
+
readonly keepPerStep?: boolean;
|
|
102
|
+
}
|
|
103
|
+
/**
|
|
104
|
+
* Survey many sessions at once.
|
|
105
|
+
*
|
|
106
|
+
* Aggregates rather than averaging the per-session percentages: a 1123-step
|
|
107
|
+
* session and a 3-step session must not count equally, and taking a mean of
|
|
108
|
+
* ratios would let the many short cheap sessions drown the few long expensive
|
|
109
|
+
* ones that carry the finding.
|
|
110
|
+
*/
|
|
111
|
+
export declare function survey(sessions: readonly HostSession[], options?: SurveyOptions): Survey;
|
|
112
|
+
/**
|
|
113
|
+
* How large a bounded prompt may be before it stops being the cheaper option.
|
|
114
|
+
*
|
|
115
|
+
* The honest way to state the finding: not "a bounded prompt saves 99%", but
|
|
116
|
+
* "a bounded prompt of up to N tokens per step is cheaper in EVERY session
|
|
117
|
+
* measured". N is a property of the data, not a number chosen to look good.
|
|
118
|
+
*/
|
|
119
|
+
export declare function breakEvenPromptTokens(sessions: readonly HostSession[]): {
|
|
120
|
+
/** Tokens per step below which bounded wins in every session. */
|
|
121
|
+
readonly tokens: number;
|
|
122
|
+
/** True when that bound came from the worst session, not the median. */
|
|
123
|
+
readonly conservative: boolean;
|
|
124
|
+
/** The average host prompt per step, for context. */
|
|
125
|
+
readonly medianAverage: number;
|
|
126
|
+
};
|
|
127
|
+
/** Render a survey as a human-readable block. */
|
|
128
|
+
export declare function formatSurvey(surveyResult: Survey): string;
|
|
129
|
+
export { assessReconstruction, reconstruct };
|
|
130
|
+
export type { HostSession, ReconstructResult, StepUsage };
|
|
131
|
+
//# sourceMappingURL=survey.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"survey.d.ts","sourceRoot":"","sources":["../../src/ab/survey.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAiCG;AAEH,OAAO,EAAE,oBAAoB,EAAE,WAAW,EAAE,MAAM,aAAa,CAAC;AAChE,OAAO,KAAK,EAAE,WAAW,EAAE,iBAAiB,EAAE,SAAS,EAAE,MAAM,aAAa,CAAC;AAE7E,iDAAiD;AACjD,MAAM,WAAW,MAAM;IACrB,oDAAoD;IACpD,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,wCAAwC;IACxC,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,uDAAuD;IACvD,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,uDAAuD;IACvD,QAAQ,CAAC,mBAAmB,EAAE,MAAM,CAAC;IACrC,gCAAgC;IAChC,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAC7B,0FAA0F;IAC1F,QAAQ,CAAC,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IACtC,sDAAsD;IACtD,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,2CAA2C;IAC3C,QAAQ,CAAC,eAAe,EAAE,MAAM,CAAC;IACjC,mDAAmD;IACnD,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B;;;;;;OAMG;IACH,QAAQ,CAAC,mBAAmB,EAAE,MAAM,CAAC;IACrC;;;OAGG;IACH,QAAQ,CAAC,sBAAsB,EAAE,MAAM,GAAG,IAAI,CAAC;IAC/C,kDAAkD;IAClD,QAAQ,CAAC,UAAU,EAAE,SAAS,MAAM,EAAE,CAAC;IACvC,iCAAiC;IACjC,QAAQ,CAAC,mBAAmB,EAAE,MAAM,CAAC;IACrC,8DAA8D;IAC9D,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B;;;;;;;;;;;;;OAaG;IACH,QAAQ,CAAC,iBAAiB,EAAE,MAAM,CAAC;IACnC,+DAA+D;IAC/D,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;IAC/B,iDAAiD;IACjD,QAAQ,CAAC,OAAO,EAAE,SAAS,iBAAiB,EAAE,CAAC;CAChD;AAED,kCAAkC;AAClC,MAAM,WAAW,aAAa;IAC5B,mEAAmE;IACnE,QAAQ,CAAC,mBAAmB,CAAC,EAAE,MAAM,CAAC;IACtC,gFAAgF;IAChF,QAAQ,CAAC,WAAW,CAAC,EAAE,OAAO,CAAC;CAChC;AAED;;;;;;;GAOG;AACH,wBAAgB,MAAM,CACpB,QAAQ,EAAE,SAAS,WAAW,EAAE,EAChC,OAAO,GAAE,aAAkB,GAC1B,MAAM,CAyDR;AAED;;;;;;GAMG;AACH,wBAAgB,qBAAqB,CAAC,QAAQ,EAAE,SAAS,WAAW,EAAE,GAAG;IACvE,iEAAiE;IACjE,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,wEAAwE;IACxE,QAAQ,CAAC,YAAY,EAAE,OAAO,CAAC;IAC/B,qDAAqD;IACrD,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;CAChC,CAiBA;AAED,iDAAiD;AACjD,wBAAgB,YAAY,CAAC,YAAY,EAAE,MAAM,GAAG,MAAM,CAuBzD;AAED,OAAO,EAAE,oBAAoB,EAAE,WAAW,EAAE,CAAC;AAC7C,YAAY,EAAE,WAAW,EAAE,iBAAiB,EAAE,SAAS,EAAE,CAAC"}
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A survey over a real host's own session store.
|
|
3
|
+
*
|
|
4
|
+
* ── What this measures, and what it does not ──────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* It answers ONE question, on real data, with no model in the loop:
|
|
7
|
+
*
|
|
8
|
+
* *given a real session's real token accounting, what would a bounded
|
|
9
|
+
* Aₜ = (P, Σₜ, Oₜ) prompt have cost on those same steps?*
|
|
10
|
+
*
|
|
11
|
+
* It does NOT answer whether an agent given that bounded prompt still does the
|
|
12
|
+
* work. That is an outcome question, it needs a live model, and no amount of
|
|
13
|
+
* offline arithmetic substitutes for it. A survey that reported a saving as if
|
|
14
|
+
* it established the claim would be repeating the original error with extra
|
|
15
|
+
* steps — the original A/B reported a number for a run whose integration had
|
|
16
|
+
* never engaged, and a cost-only win with no task completion is worth nothing.
|
|
17
|
+
*
|
|
18
|
+
* The two halves are complementary: this establishes that the cost side is
|
|
19
|
+
* real and large, the A/B establishes that the work still gets done.
|
|
20
|
+
*
|
|
21
|
+
* ── Why the host's own store ─────────────────────────────────────────────
|
|
22
|
+
*
|
|
23
|
+
* OpenCode records per-message `tokens` — `input`, `cache.read`, `output`,
|
|
24
|
+
* `reasoning`. Reading them means the numbers are the host's accounting for a
|
|
25
|
+
* real session rather than our reconstruction of it, and it works with no live
|
|
26
|
+
* model, which is what makes it usable when a provider quota is exhausted.
|
|
27
|
+
*
|
|
28
|
+
* `cache.read` is the field that carries the finding. A growing transcript is
|
|
29
|
+
* not merely expensive to send: the host caches the prefix, so re-reading
|
|
30
|
+
* history is billed as cache reads. Counting only fresh `input` would report a
|
|
31
|
+
* few percent and miss the entire effect.
|
|
32
|
+
*
|
|
33
|
+
* @non-paper — measurement infrastructure, not the paper's evaluation.
|
|
34
|
+
*/
|
|
35
|
+
import { assessReconstruction, reconstruct } from './replay.js';
|
|
36
|
+
/**
|
|
37
|
+
* Survey many sessions at once.
|
|
38
|
+
*
|
|
39
|
+
* Aggregates rather than averaging the per-session percentages: a 1123-step
|
|
40
|
+
* session and a 3-step session must not count equally, and taking a mean of
|
|
41
|
+
* ratios would let the many short cheap sessions drown the few long expensive
|
|
42
|
+
* ones that carry the finding.
|
|
43
|
+
*/
|
|
44
|
+
export function survey(sessions, options = {}) {
|
|
45
|
+
const results = [];
|
|
46
|
+
for (const session of sessions) {
|
|
47
|
+
const result = reconstruct(session, {
|
|
48
|
+
...(options.boundedPromptTokens === undefined
|
|
49
|
+
? {}
|
|
50
|
+
: { boundedPromptTokens: options.boundedPromptTokens }),
|
|
51
|
+
});
|
|
52
|
+
results.push(options.keepPerStep === true
|
|
53
|
+
? result
|
|
54
|
+
: { ...result, hostPerStep: [], boundedPerStep: [] });
|
|
55
|
+
}
|
|
56
|
+
const hostPromptTokens = results.reduce((sum, r) => sum + r.hostPromptTokens, 0);
|
|
57
|
+
const boundedPromptTokens = results.reduce((sum, r) => sum + r.boundedPromptTokens, 0);
|
|
58
|
+
const savedTokens = hostPromptTokens - boundedPromptTokens;
|
|
59
|
+
const freshInputTokens = results.reduce((sum, r) => sum + r.freshInputTokens, 0);
|
|
60
|
+
const cacheReadTokens = results.reduce((sum, r) => sum + r.cacheReadTokens, 0);
|
|
61
|
+
const hostEffectiveTokens = results.reduce((sum, r) => sum + r.hostEffectiveTokens, 0);
|
|
62
|
+
const perSession = results
|
|
63
|
+
.map((r) => r.savedFraction)
|
|
64
|
+
.filter((f) => f !== null)
|
|
65
|
+
.sort((a, b) => a - b);
|
|
66
|
+
const middle = perSession.length >> 1;
|
|
67
|
+
const medianSavedFraction = perSession.length === 0
|
|
68
|
+
? 0
|
|
69
|
+
: perSession.length % 2 === 1
|
|
70
|
+
? perSession[middle]
|
|
71
|
+
: (perSession[middle - 1] + perSession[middle]) / 2;
|
|
72
|
+
return {
|
|
73
|
+
sessions: results.length,
|
|
74
|
+
steps: results.reduce((sum, r) => sum + r.steps, 0),
|
|
75
|
+
hostPromptTokens,
|
|
76
|
+
boundedPromptTokens,
|
|
77
|
+
savedTokens,
|
|
78
|
+
savedFraction: hostPromptTokens === 0 ? null : savedTokens / hostPromptTokens,
|
|
79
|
+
freshInputTokens,
|
|
80
|
+
cacheReadTokens,
|
|
81
|
+
cacheShare: hostPromptTokens === 0 ? 0 : cacheReadTokens / hostPromptTokens,
|
|
82
|
+
hostEffectiveTokens,
|
|
83
|
+
savedEffectiveFraction: hostEffectiveTokens === 0
|
|
84
|
+
? null
|
|
85
|
+
: (hostEffectiveTokens - boundedPromptTokens) / hostEffectiveTokens,
|
|
86
|
+
perSession,
|
|
87
|
+
medianSavedFraction,
|
|
88
|
+
grewCount: results.filter((r) => r.hostSlope > 0).length,
|
|
89
|
+
boundedLosesCount: results.filter((r) => r.savedTokens < 0).length,
|
|
90
|
+
spentSessions: results.filter((r) => r.hostPromptTokens > 0).length,
|
|
91
|
+
results: [...results].sort((a, b) => b.savedTokens - a.savedTokens),
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
/**
|
|
95
|
+
* How large a bounded prompt may be before it stops being the cheaper option.
|
|
96
|
+
*
|
|
97
|
+
* The honest way to state the finding: not "a bounded prompt saves 99%", but
|
|
98
|
+
* "a bounded prompt of up to N tokens per step is cheaper in EVERY session
|
|
99
|
+
* measured". N is a property of the data, not a number chosen to look good.
|
|
100
|
+
*/
|
|
101
|
+
export function breakEvenPromptTokens(sessions) {
|
|
102
|
+
const averages = sessions
|
|
103
|
+
.filter((s) => s.steps.length > 0)
|
|
104
|
+
.map((s) => s.steps.reduce((sum, step) => sum + step.input + step.cacheRead, 0) / s.steps.length)
|
|
105
|
+
.sort((a, b) => a - b);
|
|
106
|
+
if (averages.length === 0) {
|
|
107
|
+
return { tokens: 0, conservative: true, medianAverage: 0 };
|
|
108
|
+
}
|
|
109
|
+
const middle = averages.length >> 1;
|
|
110
|
+
return {
|
|
111
|
+
tokens: Math.floor(averages[0]),
|
|
112
|
+
conservative: true,
|
|
113
|
+
medianAverage: averages.length % 2 === 1
|
|
114
|
+
? averages[middle]
|
|
115
|
+
: (averages[middle - 1] + averages[middle]) / 2,
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
/** Render a survey as a human-readable block. */
|
|
119
|
+
export function formatSurvey(surveyResult) {
|
|
120
|
+
const pct = (value) => value === null ? 'n/a' : `${(value * 100).toFixed(1)}%`;
|
|
121
|
+
return [
|
|
122
|
+
`sessions : ${surveyResult.sessions}`,
|
|
123
|
+
`steps total : ${surveyResult.steps.toLocaleString('en-US')}`,
|
|
124
|
+
'',
|
|
125
|
+
' raw token count (inflated - counts a cache read as a fresh input):',
|
|
126
|
+
` fresh input : ${surveyResult.freshInputTokens.toLocaleString('en-US')}`,
|
|
127
|
+
` cache read : ${surveyResult.cacheReadTokens.toLocaleString('en-US')} (${(surveyResult.cacheShare * 100).toFixed(1)}% of prompts)`,
|
|
128
|
+
` host total : ${surveyResult.hostPromptTokens.toLocaleString('en-US')}`,
|
|
129
|
+
` bounded A_t : ${surveyResult.boundedPromptTokens.toLocaleString('en-US')}`,
|
|
130
|
+
` raw saving : ${pct(surveyResult.savedFraction)}`,
|
|
131
|
+
'',
|
|
132
|
+
' priced in input-equivalent tokens (cache reads discounted) - QUOTE THIS:',
|
|
133
|
+
` host effective : ${Math.round(surveyResult.hostEffectiveTokens).toLocaleString('en-US')}`,
|
|
134
|
+
` bounded A_t : ${surveyResult.boundedPromptTokens.toLocaleString('en-US')}`,
|
|
135
|
+
` real saving : ${pct(surveyResult.savedEffectiveFraction)}`,
|
|
136
|
+
'',
|
|
137
|
+
`median per session : ${pct(surveyResult.medianSavedFraction)}`,
|
|
138
|
+
`transcript grew : ${surveyResult.grewCount}/${surveyResult.sessions}`,
|
|
139
|
+
`bounded loses : ${surveyResult.boundedLosesCount}/${surveyResult.spentSessions} real runs`,
|
|
140
|
+
].join('\n');
|
|
141
|
+
}
|
|
142
|
+
export { assessReconstruction, reconstruct };
|
|
143
|
+
//# sourceMappingURL=survey.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"survey.js","sourceRoot":"","sources":["../../src/ab/survey.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAiCG;AAEH,OAAO,EAAE,oBAAoB,EAAE,WAAW,EAAE,MAAM,aAAa,CAAC;AAuEhE;;;;;;;GAOG;AACH,MAAM,UAAU,MAAM,CACpB,QAAgC,EAChC,OAAO,GAAkB,EAAE;IAE3B,MAAM,OAAO,GAAwB,EAAE,CAAC;IACxC,KAAK,MAAM,OAAO,IAAI,QAAQ,EAAE,CAAC;QAC/B,MAAM,MAAM,GAAG,WAAW,CAAC,OAAO,EAAE;YAClC,GAAG,CAAC,OAAO,CAAC,mBAAmB,KAAK,SAAS;gBAC3C,CAAC,CAAC,EAAE;gBACJ,CAAC,CAAC,EAAE,mBAAmB,EAAE,OAAO,CAAC,mBAAmB,EAAE,CAAC;SAC1D,CAAC,CAAC;QACH,OAAO,CAAC,IAAI,CACV,OAAO,CAAC,WAAW,KAAK,IAAI;YAC1B,CAAC,CAAC,MAAM;YACR,CAAC,CAAC,EAAE,GAAG,MAAM,EAAE,WAAW,EAAE,EAAE,EAAE,cAAc,EAAE,EAAE,EAAE,CACvD,CAAC;IACJ,CAAC;IAED,MAAM,gBAAgB,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,CAAC,gBAAgB,EAAE,CAAC,CAAC,CAAC;IACjF,MAAM,mBAAmB,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,CAAC,mBAAmB,EAAE,CAAC,CAAC,CAAC;IACvF,MAAM,WAAW,GAAG,gBAAgB,GAAG,mBAAmB,CAAC;IAC3D,MAAM,gBAAgB,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,CAAC,gBAAgB,EAAE,CAAC,CAAC,CAAC;IACjF,MAAM,eAAe,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,CAAC,eAAe,EAAE,CAAC,CAAC,CAAC;IAC/E,MAAM,mBAAmB,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,CAAC,mBAAmB,EAAE,CAAC,CAAC,CAAC;IACvF,MAAM,UAAU,GAAG,OAAO;SACvB,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,aAAa,CAAC;SAC3B,MAAM,CAAC,CAAC,CAAC,EAAe,EAAE,CAAC,CAAC,KAAK,IAAI,CAAC;SACtC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IACzB,MAAM,MAAM,GAAG,UAAU,CAAC,MAAM,IAAI,CAAC,CAAC;IACtC,MAAM,mBAAmB,GACvB,UAAU,CAAC,MAAM,KAAK,CAAC;QACrB,CAAC,CAAC,CAAC;QACH,CAAC,CAAC,UAAU,CAAC,MAAM,GAAG,CAAC,KAAK,CAAC;YAC3B,CAAC,CAAC,UAAU,CAAC,MAAM,CAAE;YACrB,CAAC,CAAC,CAAC,UAAU,CAAC,MAAM,GAAG,CAAC,CAAE,GAAG,UAAU,CAAC,MAAM,CAAE,CAAC,GAAG,CAAC,CAAC;IAE5D,OAAO;QACL,QAAQ,EAAE,OAAO,CAAC,MAAM;QACxB,KAAK,EAAE,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,CAAC,KAAK,EAAE,CAAC,CAAC;QACnD,gBAAgB;QAChB,mBAAmB;QACnB,WAAW;QACX,aAAa,EAAE,gBAAgB,KAAK,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,WAAW,GAAG,gBAAgB;QAC7E,gBAAgB;QAChB,eAAe;QACf,UAAU,EAAE,gBAAgB,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,eAAe,GAAG,gBAAgB;QAC3E,mBAAmB;QACnB,sBAAsB,EACpB,mBAAmB,KAAK,CAAC;YACvB,CAAC,CAAC,IAAI;YACN,CAAC,CAAC,CAAC,mBAAmB,GAAG,mBAAmB,CAAC,GAAG,mBAAmB;QACvE,UAAU;QACV,mBAAmB;QACnB,SAAS,EAAE,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,SAAS,GAAG,CAAC,CAAC,CAAC,MAAM;QACxD,iBAAiB,EAAE,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,WAAW,GAAG,CAAC,CAAC,CAAC,MAAM;QAClE,aAAa,EAAE,OAAO,CAAC,MAAM,CAC3B,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,gBAAgB,GAAG,CAAC,CAC9B,CAAC,MAAM;QACR,OAAO,EAAE,CAAC,GAAG,OAAO,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,WAAW,GAAG,CAAC,CAAC,WAAW,CAAC;KACpE,CAAC;AACJ,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,qBAAqB,CAAC,QAAgC;IAQpE,MAAM,QAAQ,GAAG,QAAQ;SACtB,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC;SACjC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,IAAI,EAAE,EAAE,CAAC,GAAG,GAAG,IAAI,CAAC,KAAK,GAAG,IAAI,CAAC,SAAS,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,KAAK,CAAC,MAAM,CAAC;SAChG,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IACzB,IAAI,QAAQ,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QAC1B,OAAO,EAAE,MAAM,EAAE,CAAC,EAAE,YAAY,EAAE,IAAI,EAAE,aAAa,EAAE,CAAC,EAAE,CAAC;IAC7D,CAAC;IACD,MAAM,MAAM,GAAG,QAAQ,CAAC,MAAM,IAAI,CAAC,CAAC;IACpC,OAAO;QACL,MAAM,EAAE,IAAI,CAAC,KAAK,CAAC,QAAQ,CAAC,CAAC,CAAE,CAAC;QAChC,YAAY,EAAE,IAAI;QAClB,aAAa,EACX,QAAQ,CAAC,MAAM,GAAG,CAAC,KAAK,CAAC;YACvB,CAAC,CAAC,QAAQ,CAAC,MAAM,CAAE;YACnB,CAAC,CAAC,CAAC,QAAQ,CAAC,MAAM,GAAG,CAAC,CAAE,GAAG,QAAQ,CAAC,MAAM,CAAE,CAAC,GAAG,CAAC;KACtD,CAAC;AACJ,CAAC;AAED,iDAAiD;AACjD,MAAM,UAAU,YAAY,CAAC,YAAoB;IAC/C,MAAM,GAAG,GAAG,CAAC,KAAoB,EAAU,EAAE,CAC3C,KAAK,KAAK,IAAI,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,GAAG,CAAC,KAAK,GAAG,GAAG,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,GAAG,CAAC;IAC1D,OAAO;QACL,yBAAyB,YAAY,CAAC,QAAQ,EAAE;QAChD,yBAAyB,YAAY,CAAC,KAAK,CAAC,cAAc,CAAC,OAAO,CAAC,EAAE;QACrE,EAAE;QACF,sEAAsE;QACtE,0BAA0B,YAAY,CAAC,gBAAgB,CAAC,cAAc,CAAC,OAAO,CAAC,EAAE;QACjF,0BAA0B,YAAY,CAAC,eAAe,CAAC,cAAc,CAAC,OAAO,CAAC,KAAK,CAAC,YAAY,CAAC,UAAU,GAAG,GAAG,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,eAAe;QAC5I,0BAA0B,YAAY,CAAC,gBAAgB,CAAC,cAAc,CAAC,OAAO,CAAC,EAAE;QACjF,0BAA0B,YAAY,CAAC,mBAAmB,CAAC,cAAc,CAAC,OAAO,CAAC,EAAE;QACpF,0BAA0B,GAAG,CAAC,YAAY,CAAC,aAAa,CAAC,EAAE;QAC3D,EAAE;QACF,4EAA4E;QAC5E,0BAA0B,IAAI,CAAC,KAAK,CAAC,YAAY,CAAC,mBAAmB,CAAC,CAAC,cAAc,CAAC,OAAO,CAAC,EAAE;QAChG,0BAA0B,YAAY,CAAC,mBAAmB,CAAC,cAAc,CAAC,OAAO,CAAC,EAAE;QACpF,0BAA0B,GAAG,CAAC,YAAY,CAAC,sBAAsB,CAAC,EAAE;QACpE,EAAE;QACF,yBAAyB,GAAG,CAAC,YAAY,CAAC,mBAAmB,CAAC,EAAE;QAChE,yBAAyB,YAAY,CAAC,SAAS,IAAI,YAAY,CAAC,QAAQ,EAAE;QAC1E,yBAAyB,YAAY,CAAC,iBAAiB,IAAI,YAAY,CAAC,aAAa,YAAY;KAClG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AACf,CAAC;AAED,OAAO,EAAE,oBAAoB,EAAE,WAAW,EAAE,CAAC"}
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Token accounting read from the HOST's own store.
|
|
3
|
+
*
|
|
4
|
+
* ── Why not parse the host's stdout ──────────────────────────────────────
|
|
5
|
+
*
|
|
6
|
+
* Three reasons, in order of how much they matter:
|
|
7
|
+
*
|
|
8
|
+
* 1. **The model does not report its own cost reliably.** A harness that
|
|
9
|
+
* takes the number the subject of the experiment wrote down is trusting the
|
|
10
|
+
* measurement to the thing being measured. The previous A/B took its
|
|
11
|
+
* figures from the run's own reporting, and one of its two "arms" was
|
|
12
|
+
* itself two different readings of a single session.
|
|
13
|
+
* 2. **OpenCode already records per-message token usage** in its local store
|
|
14
|
+
* (`message.data.tokens` with `input`, `output`, `reasoning` and
|
|
15
|
+
* `cache.{read,write}`). That is the host's own accounting, not a
|
|
16
|
+
* reconstruction, and it is keyed by session so a run can be re-read
|
|
17
|
+
* after the fact.
|
|
18
|
+
* 3. **It is available without a live model.** Reading the store works when
|
|
19
|
+
* the provider quota is exhausted, which is exactly when one most wants to
|
|
20
|
+
* re-analyse a previous run.
|
|
21
|
+
*
|
|
22
|
+
* Node's `node:sqlite` is used rather than a dependency: the store is a
|
|
23
|
+
* local file, the read is a single prepared statement, and adding a native
|
|
24
|
+
* dependency to a zero-deps package for this would be a poor trade. The
|
|
25
|
+
* reader is injected, so the query and its failure modes are testable
|
|
26
|
+
* without an OpenCode installation.
|
|
27
|
+
*
|
|
28
|
+
* @non-paper — measurement infrastructure, not the paper's evaluation.
|
|
29
|
+
*/
|
|
30
|
+
import type { TokenUsage } from './record.js';
|
|
31
|
+
/** One row of the host's message table, as far as this module cares. */
|
|
32
|
+
export interface HostMessageRow {
|
|
33
|
+
/** The session the message belongs to. */
|
|
34
|
+
readonly sessionID: string;
|
|
35
|
+
/** Milliseconds since the epoch. */
|
|
36
|
+
readonly created: number;
|
|
37
|
+
/** `assistant`, `user`, or anything else this module ignores. */
|
|
38
|
+
readonly role: string;
|
|
39
|
+
/** Token accounting as the host recorded it. */
|
|
40
|
+
readonly tokens: Partial<TokenUsage> | null;
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* The narrow seam onto the host's store.
|
|
44
|
+
*
|
|
45
|
+
* Injecting this is what lets the arithmetic below be tested against
|
|
46
|
+
* hand-written rows instead of against a live OpenCode server, and what lets
|
|
47
|
+
* a caller substitute a different source (a JSON export, a CI fixture)
|
|
48
|
+
* without touching the logic that consumes it.
|
|
49
|
+
*/
|
|
50
|
+
export interface UsageReader {
|
|
51
|
+
/**
|
|
52
|
+
* Every message recorded for `sessionID`, in creation order.
|
|
53
|
+
*
|
|
54
|
+
* Rejects when the store is missing, locked, or the session is unknown —
|
|
55
|
+
* an unreadable store must surface as a failed run, never as zero tokens,
|
|
56
|
+
* because zero tokens would make an arm look free.
|
|
57
|
+
*/
|
|
58
|
+
messagesFor(sessionID: string): Promise<readonly HostMessageRow[]>;
|
|
59
|
+
}
|
|
60
|
+
/** Zero usage, for a session whose messages carry no accounting. */
|
|
61
|
+
export declare const NO_USAGE: TokenUsage;
|
|
62
|
+
/**
|
|
63
|
+
* Total token spend for a session.
|
|
64
|
+
*
|
|
65
|
+
* Only `assistant` rows count. A user message's `input` figure is the host
|
|
66
|
+
* describing the prompt it echoed, and adding it would double-count the
|
|
67
|
+
* conversation.
|
|
68
|
+
*
|
|
69
|
+
* `cacheRead` is summed, not treated as free: prompt caching changes the
|
|
70
|
+
* price of a token, not the fact that the model was shown it, and the
|
|
71
|
+
* paper's claim is about what the model is exposed to.
|
|
72
|
+
*/
|
|
73
|
+
export declare function sessionUsage(rows: readonly HostMessageRow[]): TokenUsage;
|
|
74
|
+
/** Why a usage lookup produced nothing. */
|
|
75
|
+
export type UsageFailure =
|
|
76
|
+
/** The host store could not be opened or read. */
|
|
77
|
+
'store_unavailable'
|
|
78
|
+
/** The session id is not in the host's store. */
|
|
79
|
+
| 'session_unknown';
|
|
80
|
+
/** The outcome of resolving one session's token spend. */
|
|
81
|
+
export type UsageOutcome = {
|
|
82
|
+
readonly ok: true;
|
|
83
|
+
readonly usage: TokenUsage;
|
|
84
|
+
readonly assistantMessages: number;
|
|
85
|
+
} | {
|
|
86
|
+
readonly ok: false;
|
|
87
|
+
readonly reason: UsageFailure;
|
|
88
|
+
readonly detail: string;
|
|
89
|
+
};
|
|
90
|
+
/**
|
|
91
|
+
* Resolve a session's token spend, turning every failure into a value.
|
|
92
|
+
*
|
|
93
|
+
* A run whose accounting cannot be read is a run with unknown cost, which is
|
|
94
|
+
* not the same as a run that cost nothing. The distinction is the whole
|
|
95
|
+
* reason this returns a union instead of a number: `store_unavailable` must
|
|
96
|
+
* never be silently read as `NO_USAGE`.
|
|
97
|
+
*/
|
|
98
|
+
export declare function resolveSessionUsage(reader: UsageReader, sessionID: string): Promise<UsageOutcome>;
|
|
99
|
+
//# sourceMappingURL=usage.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"usage.d.ts","sourceRoot":"","sources":["../../src/ab/usage.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA4BG;AAEH,OAAO,KAAK,EAAE,UAAU,EAAE,MAAM,aAAa,CAAC;AAE9C,wEAAwE;AACxE,MAAM,WAAW,cAAc;IAC7B,0CAA0C;IAC1C,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,oCAAoC;IACpC,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;IACzB,iEAAiE;IACjE,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAC;IACtB,gDAAgD;IAChD,QAAQ,CAAC,MAAM,EAAE,OAAO,CAAC,UAAU,CAAC,GAAG,IAAI,CAAC;CAC7C;AAED;;;;;;;GAOG;AACH,MAAM,WAAW,WAAW;IAC1B;;;;;;OAMG;IACH,WAAW,CAAC,SAAS,EAAE,MAAM,GAAG,OAAO,CAAC,SAAS,cAAc,EAAE,CAAC,CAAC;CACpE;AAED,oEAAoE;AACpE,eAAO,MAAM,QAAQ,EAAE,UAKtB,CAAC;AAYF;;;;;;;;;;GAUG;AACH,wBAAgB,YAAY,CAAC,IAAI,EAAE,SAAS,cAAc,EAAE,GAAG,UAAU,CAaxE;AAED,2CAA2C;AAC3C,MAAM,MAAM,YAAY;AACtB,kDAAkD;AAChD,mBAAmB;AACrB,iDAAiD;GAC/C,iBAAiB,CAAC;AAEtB,0DAA0D;AAC1D,MAAM,MAAM,YAAY,GACpB;IAAE,QAAQ,CAAC,EAAE,EAAE,IAAI,CAAC;IAAC,QAAQ,CAAC,KAAK,EAAE,UAAU,CAAC;IAAC,QAAQ,CAAC,iBAAiB,EAAE,MAAM,CAAA;CAAE,GACrF;IAAE,QAAQ,CAAC,EAAE,EAAE,KAAK,CAAC;IAAC,QAAQ,CAAC,MAAM,EAAE,YAAY,CAAC;IAAC,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAA;CAAE,CAAC;AAEnF;;;;;;;GAOG;AACH,wBAAsB,mBAAmB,CACvC,MAAM,EAAE,WAAW,EACnB,SAAS,EAAE,MAAM,GAChB,OAAO,CAAC,YAAY,CAAC,CAoBvB"}
|