@skillstate/bench 3.0.0 → 3.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +1 -2
  2. package/dist/ab/engagement.d.ts +70 -0
  3. package/dist/ab/engagement.d.ts.map +1 -0
  4. package/dist/ab/engagement.js +102 -0
  5. package/dist/ab/engagement.js.map +1 -0
  6. package/dist/ab/index.d.ts +28 -0
  7. package/dist/ab/index.d.ts.map +1 -0
  8. package/dist/ab/index.js +20 -0
  9. package/dist/ab/index.js.map +1 -0
  10. package/dist/ab/opencode-usage.d.ts +52 -0
  11. package/dist/ab/opencode-usage.d.ts.map +1 -0
  12. package/dist/ab/opencode-usage.js +85 -0
  13. package/dist/ab/opencode-usage.js.map +1 -0
  14. package/dist/ab/record.d.ts +173 -0
  15. package/dist/ab/record.d.ts.map +1 -0
  16. package/dist/ab/record.js +98 -0
  17. package/dist/ab/record.js.map +1 -0
  18. package/dist/ab/replay.d.ts +193 -0
  19. package/dist/ab/replay.d.ts.map +1 -0
  20. package/dist/ab/replay.js +149 -0
  21. package/dist/ab/replay.js.map +1 -0
  22. package/dist/ab/report.d.ts +26 -0
  23. package/dist/ab/report.d.ts.map +1 -0
  24. package/dist/ab/report.js +98 -0
  25. package/dist/ab/report.js.map +1 -0
  26. package/dist/ab/stats-core.d.ts +84 -0
  27. package/dist/ab/stats-core.d.ts.map +1 -0
  28. package/dist/ab/stats-core.js +93 -0
  29. package/dist/ab/stats-core.js.map +1 -0
  30. package/dist/ab/survey.d.ts +131 -0
  31. package/dist/ab/survey.d.ts.map +1 -0
  32. package/dist/ab/survey.js +143 -0
  33. package/dist/ab/survey.js.map +1 -0
  34. package/dist/ab/usage.d.ts +99 -0
  35. package/dist/ab/usage.d.ts.map +1 -0
  36. package/dist/ab/usage.js +98 -0
  37. package/dist/ab/usage.js.map +1 -0
  38. package/dist/ab/verdict.d.ts +88 -0
  39. package/dist/ab/verdict.d.ts.map +1 -0
  40. package/dist/ab/verdict.js +251 -0
  41. package/dist/ab/verdict.js.map +1 -0
  42. package/dist/ab-cli.d.ts +35 -0
  43. package/dist/ab-cli.d.ts.map +1 -0
  44. package/dist/ab-cli.js +161 -0
  45. package/dist/ab-cli.js.map +1 -0
  46. package/dist/index.d.ts +1 -0
  47. package/dist/index.d.ts.map +1 -1
  48. package/dist/index.js +1 -0
  49. package/dist/index.js.map +1 -1
  50. package/package.json +5 -1
@@ -0,0 +1,84 @@
1
+ /**
2
+ * Distribution summaries for a set of runs.
3
+ *
4
+ * ── Why the median and not the mean ──────────────────────────────────────
5
+ *
6
+ * Run-to-run variance in a non-deterministic host is large — the old A/B
7
+ * measured the same arm twice and got 42 364 and 71 590 input tokens, a 69%
8
+ * swing between a mid-run reading and the final one. A mean over a handful
9
+ * of such samples is dominated by whichever sample was luckiest, and a
10
+ * percentage computed from it inherits that luck.
11
+ *
12
+ * The median is reported instead, together with the median absolute
13
+ * deviation (MAD), so a reader sees the SPREAD next to the effect. An effect
14
+ * smaller than the spread is not an effect, and the harness in `verdict.ts`
15
+ * refuses to call it one.
16
+ *
17
+ * These are deliberately dependency-free and deterministic: same input, same
18
+ * output, no RNG anywhere. That is what lets the gates be tested against the
19
+ * real historical numbers.
20
+ *
21
+ * @non-paper — measurement infrastructure, not the paper's evaluation.
22
+ */
23
+ /**
24
+ * Median of a non-empty numeric sample.
25
+ *
26
+ * Requires a non-empty array and throws on an empty one: an empty sample has
27
+ * no median, and inventing one (0, or NaN) would let a silently-unrun arm
28
+ * compare as if it had run. The gates validate sample size before calling.
29
+ */
30
+ export declare function median(values: readonly number[]): number;
31
+ /**
32
+ * Median absolute deviation — a spread measure that ignores outliers.
33
+ *
34
+ * Subtracting the median first is what makes it robust: one runaway run
35
+ * inflates a standard deviation enough to hide a real effect, but barely
36
+ * moves the MAD, because the runaway run's own deviation is measured against
37
+ * a median that the median itself did not follow.
38
+ */
39
+ export declare function mad(values: readonly number[]): number;
40
+ /** The spread of a sample, as a fraction of its median. Zero when the median is 0. */
41
+ export declare function relativeMad(values: readonly number[]): number;
42
+ /** A sample summarised for reporting. */
43
+ export interface Distribution {
44
+ readonly n: number;
45
+ readonly min: number;
46
+ readonly median: number;
47
+ readonly max: number;
48
+ /** Median absolute deviation, in the same units as the sample. */
49
+ readonly mad: number;
50
+ /** MAD as a fraction of the median. 0 when the median is 0. */
51
+ readonly relativeMad: number;
52
+ /** The raw values, sorted ascending, so a reader can check the summary. */
53
+ readonly sorted: readonly number[];
54
+ }
55
+ /** Summarise a sample. Throws on an empty one — see {@link median}. */
56
+ export declare function describe(values: readonly number[]): Distribution;
57
+ /**
58
+ * A relative effect size, robust to scale and to spread.
59
+ *
60
+ * `effect = (control − instrumented) / pooledSpread`, where `pooledSpread` is
61
+ * the larger of the two arms' MADs. The sign carries the direction (positive
62
+ * means the instrumented arm spent fewer prompt tokens); the magnitude is the
63
+ * number of MADs the gap is worth.
64
+ *
65
+ * When both arms have zero spread the comparison is exact and the ratio is
66
+ * undefined, so `exact: true` is returned with the plain difference. Callers
67
+ * must branch on `exact` rather than dividing by the missing ratio — that
68
+ * branch is where a harness quietly invents an effect.
69
+ */
70
+ export interface EffectSize {
71
+ /** False when either arm had zero spread and the ratio cannot be formed. */
72
+ readonly exact: boolean;
73
+ /** `control − instrumented`, in tokens. Positive favours the instrumented arm. */
74
+ readonly difference: number;
75
+ /** Difference as a fraction of the control median. Null when the median is 0. */
76
+ readonly relative: number | null;
77
+ /** Difference in units of pooled MAD. Null when `exact` is false. */
78
+ readonly inMads: number | null;
79
+ }
80
+ /** Minimum effect, in MADs, the harness is willing to call a signal. */
81
+ export declare const MIN_EFFECT_IN_MADS = 1;
82
+ /** Compare two arms' prompt-token distributions. Throws if either sample is empty. */
83
+ export declare function effectSize(control: readonly number[], instrumented: readonly number[]): EffectSize;
84
+ //# sourceMappingURL=stats-core.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"stats-core.d.ts","sourceRoot":"","sources":["../../src/ab/stats-core.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH;;;;;;GAMG;AACH,wBAAgB,MAAM,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,CASxD;AAED;;;;;;;GAOG;AACH,wBAAgB,GAAG,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,CAMrD;AAED,sFAAsF;AACtF,wBAAgB,WAAW,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,CAG7D;AAED,yCAAyC;AACzC,MAAM,WAAW,YAAY;IAC3B,QAAQ,CAAC,CAAC,EAAE,MAAM,CAAC;IACnB,QAAQ,CAAC,GAAG,EAAE,MAAM,CAAC;IACrB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,QAAQ,CAAC,GAAG,EAAE,MAAM,CAAC;IACrB,kEAAkE;IAClE,QAAQ,CAAC,GAAG,EAAE,MAAM,CAAC;IACrB,+DAA+D;IAC/D,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAC7B,2EAA2E;IAC3E,QAAQ,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,CAAC;CACpC;AAED,uEAAuE;AACvE,wBAAgB,QAAQ,CAAC,MAAM,EAAE,SAAS,MAAM,EAAE,GAAG,YAAY,CAchE;AAED;;;;;;;;;;;;GAYG;AACH,MAAM,WAAW,UAAU;IACzB,4EAA4E;IAC5E,QAAQ,CAAC,KAAK,EAAE,OAAO,CAAC;IACxB,kFAAkF;IAClF,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B,iFAAiF;IACjF,QAAQ,CAAC,QAAQ,EAAE,MAAM,GAAG,IAAI,CAAC;IACjC,qEAAqE;IACrE,QAAQ,CAAC,MAAM,EAAE,MAAM,GAAG,IAAI,CAAC;CAChC;AAED,wEAAwE;AACxE,eAAO,MAAM,kBAAkB,IAAI,CAAC;AAEpC,sFAAsF;AACtF,wBAAgB,UAAU,CACxB,OAAO,EAAE,SAAS,MAAM,EAAE,EAC1B,YAAY,EAAE,SAAS,MAAM,EAAE,GAC9B,UAAU,CAcZ"}
@@ -0,0 +1,93 @@
1
+ /**
2
+ * Distribution summaries for a set of runs.
3
+ *
4
+ * ── Why the median and not the mean ──────────────────────────────────────
5
+ *
6
+ * Run-to-run variance in a non-deterministic host is large — the old A/B
7
+ * measured the same arm twice and got 42 364 and 71 590 input tokens, a 69%
8
+ * swing between a mid-run reading and the final one. A mean over a handful
9
+ * of such samples is dominated by whichever sample was luckiest, and a
10
+ * percentage computed from it inherits that luck.
11
+ *
12
+ * The median is reported instead, together with the median absolute
13
+ * deviation (MAD), so a reader sees the SPREAD next to the effect. An effect
14
+ * smaller than the spread is not an effect, and the harness in `verdict.ts`
15
+ * refuses to call it one.
16
+ *
17
+ * These are deliberately dependency-free and deterministic: same input, same
18
+ * output, no RNG anywhere. That is what lets the gates be tested against the
19
+ * real historical numbers.
20
+ *
21
+ * @non-paper — measurement infrastructure, not the paper's evaluation.
22
+ */
23
+ /**
24
+ * Median of a non-empty numeric sample.
25
+ *
26
+ * Requires a non-empty array and throws on an empty one: an empty sample has
27
+ * no median, and inventing one (0, or NaN) would let a silently-unrun arm
28
+ * compare as if it had run. The gates validate sample size before calling.
29
+ */
30
+ export function median(values) {
31
+ if (values.length === 0) {
32
+ throw new RangeError('median of an empty sample is undefined');
33
+ }
34
+ const sorted = [...values].sort((a, b) => a - b);
35
+ const middle = sorted.length >> 1;
36
+ return sorted.length % 2 === 1
37
+ ? sorted[middle]
38
+ : (sorted[middle - 1] + sorted[middle]) / 2;
39
+ }
40
+ /**
41
+ * Median absolute deviation — a spread measure that ignores outliers.
42
+ *
43
+ * Subtracting the median first is what makes it robust: one runaway run
44
+ * inflates a standard deviation enough to hide a real effect, but barely
45
+ * moves the MAD, because the runaway run's own deviation is measured against
46
+ * a median that the median itself did not follow.
47
+ */
48
+ export function mad(values) {
49
+ if (values.length === 0) {
50
+ throw new RangeError('MAD of an empty sample is undefined');
51
+ }
52
+ const center = median(values);
53
+ return median(values.map((value) => Math.abs(value - center)));
54
+ }
55
+ /** The spread of a sample, as a fraction of its median. Zero when the median is 0. */
56
+ export function relativeMad(values) {
57
+ const center = median(values);
58
+ return center === 0 ? 0 : mad(values) / center;
59
+ }
60
+ /** Summarise a sample. Throws on an empty one — see {@link median}. */
61
+ export function describe(values) {
62
+ if (values.length === 0) {
63
+ throw new RangeError('cannot describe an empty sample');
64
+ }
65
+ const sorted = [...values].sort((a, b) => a - b);
66
+ return {
67
+ n: values.length,
68
+ min: sorted[0],
69
+ median: median(values),
70
+ max: sorted[sorted.length - 1],
71
+ mad: mad(values),
72
+ relativeMad: relativeMad(values),
73
+ sorted,
74
+ };
75
+ }
76
+ /** Minimum effect, in MADs, the harness is willing to call a signal. */
77
+ export const MIN_EFFECT_IN_MADS = 1;
78
+ /** Compare two arms' prompt-token distributions. Throws if either sample is empty. */
79
+ export function effectSize(control, instrumented) {
80
+ if (control.length === 0 || instrumented.length === 0) {
81
+ throw new RangeError('effect size needs a non-empty sample in both arms');
82
+ }
83
+ const controlMedian = median(control);
84
+ const instrumentedMedian = median(instrumented);
85
+ const difference = controlMedian - instrumentedMedian;
86
+ const relative = controlMedian === 0 ? null : difference / controlMedian;
87
+ const spread = Math.max(mad(control), mad(instrumented));
88
+ if (spread === 0) {
89
+ return { exact: true, difference, relative, inMads: null };
90
+ }
91
+ return { exact: false, difference, relative, inMads: difference / spread };
92
+ }
93
+ //# sourceMappingURL=stats-core.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"stats-core.js","sourceRoot":"","sources":["../../src/ab/stats-core.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH;;;;;;GAMG;AACH,MAAM,UAAU,MAAM,CAAC,MAAyB;IAC9C,IAAI,MAAM,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACxB,MAAM,IAAI,UAAU,CAAC,wCAAwC,CAAC,CAAC;IACjE,CAAC;IACD,MAAM,MAAM,GAAG,CAAC,GAAG,MAAM,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IACjD,MAAM,MAAM,GAAG,MAAM,CAAC,MAAM,IAAI,CAAC,CAAC;IAClC,OAAO,MAAM,CAAC,MAAM,GAAG,CAAC,KAAK,CAAC;QAC5B,CAAC,CAAC,MAAM,CAAC,MAAM,CAAE;QACjB,CAAC,CAAC,CAAC,MAAM,CAAC,MAAM,GAAG,CAAC,CAAE,GAAG,MAAM,CAAC,MAAM,CAAE,CAAC,GAAG,CAAC,CAAC;AAClD,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,GAAG,CAAC,MAAyB;IAC3C,IAAI,MAAM,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACxB,MAAM,IAAI,UAAU,CAAC,qCAAqC,CAAC,CAAC;IAC9D,CAAC;IACD,MAAM,MAAM,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC;IAC9B,OAAO,MAAM,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,GAAG,MAAM,CAAC,CAAC,CAAC,CAAC;AACjE,CAAC;AAED,sFAAsF;AACtF,MAAM,UAAU,WAAW,CAAC,MAAyB;IACnD,MAAM,MAAM,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC;IAC9B,OAAO,MAAM,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,MAAM,CAAC,GAAG,MAAM,CAAC;AACjD,CAAC;AAgBD,uEAAuE;AACvE,MAAM,UAAU,QAAQ,CAAC,MAAyB;IAChD,IAAI,MAAM,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACxB,MAAM,IAAI,UAAU,CAAC,iCAAiC,CAAC,CAAC;IAC1D,CAAC;IACD,MAAM,MAAM,GAAG,CAAC,GAAG,MAAM,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IACjD,OAAO;QACL,CAAC,EAAE,MAAM,CAAC,MAAM;QAChB,GAAG,EAAE,MAAM,CAAC,CAAC,CAAE;QACf,MAAM,EAAE,MAAM,CAAC,MAAM,CAAC;QACtB,GAAG,EAAE,MAAM,CAAC,MAAM,CAAC,MAAM,GAAG,CAAC,CAAE;QAC/B,GAAG,EAAE,GAAG,CAAC,MAAM,CAAC;QAChB,WAAW,EAAE,WAAW,CAAC,MAAM,CAAC;QAChC,MAAM;KACP,CAAC;AACJ,CAAC;AA0BD,wEAAwE;AACxE,MAAM,CAAC,MAAM,kBAAkB,GAAG,CAAC,CAAC;AAEpC,sFAAsF;AACtF,MAAM,UAAU,UAAU,CACxB,OAA0B,EAC1B,YAA+B;IAE/B,IAAI,OAAO,CAAC,MAAM,KAAK,CAAC,IAAI,YAAY,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACtD,MAAM,IAAI,UAAU,CAAC,mDAAmD,CAAC,CAAC;IAC5E,CAAC;IACD,MAAM,aAAa,GAAG,MAAM,CAAC,OAAO,CAAC,CAAC;IACtC,MAAM,kBAAkB,GAAG,MAAM,CAAC,YAAY,CAAC,CAAC;IAChD,MAAM,UAAU,GAAG,aAAa,GAAG,kBAAkB,CAAC;IACtD,MAAM,QAAQ,GAAG,aAAa,KAAK,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,UAAU,GAAG,aAAa,CAAC;IACzE,MAAM,MAAM,GAAG,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,OAAO,CAAC,EAAE,GAAG,CAAC,YAAY,CAAC,CAAC,CAAC;IAEzD,IAAI,MAAM,KAAK,CAAC,EAAE,CAAC;QACjB,OAAO,EAAE,KAAK,EAAE,IAAI,EAAE,UAAU,EAAE,QAAQ,EAAE,MAAM,EAAE,IAAI,EAAE,CAAC;IAC7D,CAAC;IACD,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,UAAU,EAAE,QAAQ,EAAE,MAAM,EAAE,UAAU,GAAG,MAAM,EAAE,CAAC;AAC7E,CAAC"}
@@ -0,0 +1,131 @@
1
+ /**
2
+ * A survey over a real host's own session store.
3
+ *
4
+ * ── What this measures, and what it does not ──────────────────────────────
5
+ *
6
+ * It answers ONE question, on real data, with no model in the loop:
7
+ *
8
+ * *given a real session's real token accounting, what would a bounded
9
+ * Aₜ = (P, Σₜ, Oₜ) prompt have cost on those same steps?*
10
+ *
11
+ * It does NOT answer whether an agent given that bounded prompt still does the
12
+ * work. That is an outcome question, it needs a live model, and no amount of
13
+ * offline arithmetic substitutes for it. A survey that reported a saving as if
14
+ * it established the claim would be repeating the original error with extra
15
+ * steps — the original A/B reported a number for a run whose integration had
16
+ * never engaged, and a cost-only win with no task completion is worth nothing.
17
+ *
18
+ * The two halves are complementary: this establishes that the cost side is
19
+ * real and large, the A/B establishes that the work still gets done.
20
+ *
21
+ * ── Why the host's own store ─────────────────────────────────────────────
22
+ *
23
+ * OpenCode records per-message `tokens` — `input`, `cache.read`, `output`,
24
+ * `reasoning`. Reading them means the numbers are the host's accounting for a
25
+ * real session rather than our reconstruction of it, and it works with no live
26
+ * model, which is what makes it usable when a provider quota is exhausted.
27
+ *
28
+ * `cache.read` is the field that carries the finding. A growing transcript is
29
+ * not merely expensive to send: the host caches the prefix, so re-reading
30
+ * history is billed as cache reads. Counting only fresh `input` would report a
31
+ * few percent and miss the entire effect.
32
+ *
33
+ * @non-paper — measurement infrastructure, not the paper's evaluation.
34
+ */
35
+ import { assessReconstruction, reconstruct } from './replay.js';
36
+ import type { HostSession, ReconstructResult, StepUsage } from './replay.js';
37
+ /** The result of surveying a set of sessions. */
38
+ export interface Survey {
39
+ /** Sessions with enough steps to compare curves. */
40
+ readonly sessions: number;
41
+ /** Total steps across every session. */
42
+ readonly steps: number;
43
+ /** What the host actually spent on prompts, summed. */
44
+ readonly hostPromptTokens: number;
45
+ /** What bounded Aₜ prompts would have cost, summed. */
46
+ readonly boundedPromptTokens: number;
47
+ /** `host − bounded`, summed. */
48
+ readonly savedTokens: number;
49
+ /** `saved / host`, as a raw token count. Inflated; see {@link savedEffectiveFraction}. */
50
+ readonly savedFraction: number | null;
51
+ /** Fresh, uncached input tokens across the corpus. */
52
+ readonly freshInputTokens: number;
53
+ /** Cache-read tokens across the corpus. */
54
+ readonly cacheReadTokens: number;
55
+ /** Cache reads as a share of all prompt tokens. */
56
+ readonly cacheShare: number;
57
+ /**
58
+ * The corpus priced in input-equivalent tokens, cache reads discounted.
59
+ *
60
+ * The number to quote. The raw total counts a cache read as equal to a
61
+ * fresh input token, and on a cache-heavy corpus that inflates a saving by
62
+ * roughly an order of magnitude.
63
+ */
64
+ readonly hostEffectiveTokens: number;
65
+ /**
66
+ * The saving priced properly. Always ≤ {@link savedFraction}, and the
67
+ * defensible one.
68
+ */
69
+ readonly savedEffectiveFraction: number | null;
70
+ /** Per-session saving fractions, for a median. */
71
+ readonly perSession: readonly number[];
72
+ /** Median per-session saving. */
73
+ readonly medianSavedFraction: number;
74
+ /** Sessions whose prompt grew between first and last step. */
75
+ readonly grewCount: number;
76
+ /**
77
+ * Sessions where the bounded prompt would have cost MORE.
78
+ *
79
+ * The strongest number in the survey: a bounded prompt that never loses is
80
+ * a different claim from one that usually wins, and only the counter can
81
+ * distinguish them.
82
+ *
83
+ * CAVEAT when quoting it: this counts every session, including rows that
84
+ * recorded zero tokens — aborted runs and records written before
85
+ * accounting was populated. Such a session is not "a run that came out
86
+ * cheap", it is a run that never happened, and it will register as a loss
87
+ * for any bounded prompt. Filter on {@link spentSessions} before claiming
88
+ * "never loses", or the number is smaller and more honest.
89
+ */
90
+ readonly boundedLosesCount: number;
91
+ /** Sessions where the host recorded any token spend at all. */
92
+ readonly spentSessions: number;
93
+ /** Per-session results, largest saving first. */
94
+ readonly results: readonly ReconstructResult[];
95
+ }
96
+ /** Options for {@link survey}. */
97
+ export interface SurveyOptions {
98
+ /** Tokens per bounded Aₜ prompt. Defaults to the paper's ~1.8k. */
99
+ readonly boundedPromptTokens?: number;
100
+ /** Include a session's full per-step arrays in `results`. Defaults to false. */
101
+ readonly keepPerStep?: boolean;
102
+ }
103
+ /**
104
+ * Survey many sessions at once.
105
+ *
106
+ * Aggregates rather than averaging the per-session percentages: a 1123-step
107
+ * session and a 3-step session must not count equally, and taking a mean of
108
+ * ratios would let the many short cheap sessions drown the few long expensive
109
+ * ones that carry the finding.
110
+ */
111
+ export declare function survey(sessions: readonly HostSession[], options?: SurveyOptions): Survey;
112
+ /**
113
+ * How large a bounded prompt may be before it stops being the cheaper option.
114
+ *
115
+ * The honest way to state the finding: not "a bounded prompt saves 99%", but
116
+ * "a bounded prompt of up to N tokens per step is cheaper in EVERY session
117
+ * measured". N is a property of the data, not a number chosen to look good.
118
+ */
119
+ export declare function breakEvenPromptTokens(sessions: readonly HostSession[]): {
120
+ /** Tokens per step below which bounded wins in every session. */
121
+ readonly tokens: number;
122
+ /** True when that bound came from the worst session, not the median. */
123
+ readonly conservative: boolean;
124
+ /** The average host prompt per step, for context. */
125
+ readonly medianAverage: number;
126
+ };
127
+ /** Render a survey as a human-readable block. */
128
+ export declare function formatSurvey(surveyResult: Survey): string;
129
+ export { assessReconstruction, reconstruct };
130
+ export type { HostSession, ReconstructResult, StepUsage };
131
+ //# sourceMappingURL=survey.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"survey.d.ts","sourceRoot":"","sources":["../../src/ab/survey.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAiCG;AAEH,OAAO,EAAE,oBAAoB,EAAE,WAAW,EAAE,MAAM,aAAa,CAAC;AAChE,OAAO,KAAK,EAAE,WAAW,EAAE,iBAAiB,EAAE,SAAS,EAAE,MAAM,aAAa,CAAC;AAE7E,iDAAiD;AACjD,MAAM,WAAW,MAAM;IACrB,oDAAoD;IACpD,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAC1B,wCAAwC;IACxC,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,uDAAuD;IACvD,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,uDAAuD;IACvD,QAAQ,CAAC,mBAAmB,EAAE,MAAM,CAAC;IACrC,gCAAgC;IAChC,QAAQ,CAAC,WAAW,EAAE,MAAM,CAAC;IAC7B,0FAA0F;IAC1F,QAAQ,CAAC,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IACtC,sDAAsD;IACtD,QAAQ,CAAC,gBAAgB,EAAE,MAAM,CAAC;IAClC,2CAA2C;IAC3C,QAAQ,CAAC,eAAe,EAAE,MAAM,CAAC;IACjC,mDAAmD;IACnD,QAAQ,CAAC,UAAU,EAAE,MAAM,CAAC;IAC5B;;;;;;OAMG;IACH,QAAQ,CAAC,mBAAmB,EAAE,MAAM,CAAC;IACrC;;;OAGG;IACH,QAAQ,CAAC,sBAAsB,EAAE,MAAM,GAAG,IAAI,CAAC;IAC/C,kDAAkD;IAClD,QAAQ,CAAC,UAAU,EAAE,SAAS,MAAM,EAAE,CAAC;IACvC,iCAAiC;IACjC,QAAQ,CAAC,mBAAmB,EAAE,MAAM,CAAC;IACrC,8DAA8D;IAC9D,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B;;;;;;;;;;;;;OAaG;IACH,QAAQ,CAAC,iBAAiB,EAAE,MAAM,CAAC;IACnC,+DAA+D;IAC/D,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;IAC/B,iDAAiD;IACjD,QAAQ,CAAC,OAAO,EAAE,SAAS,iBAAiB,EAAE,CAAC;CAChD;AAED,kCAAkC;AAClC,MAAM,WAAW,aAAa;IAC5B,mEAAmE;IACnE,QAAQ,CAAC,mBAAmB,CAAC,EAAE,MAAM,CAAC;IACtC,gFAAgF;IAChF,QAAQ,CAAC,WAAW,CAAC,EAAE,OAAO,CAAC;CAChC;AAED;;;;;;;GAOG;AACH,wBAAgB,MAAM,CACpB,QAAQ,EAAE,SAAS,WAAW,EAAE,EAChC,OAAO,GAAE,aAAkB,GAC1B,MAAM,CAyDR;AAED;;;;;;GAMG;AACH,wBAAgB,qBAAqB,CAAC,QAAQ,EAAE,SAAS,WAAW,EAAE,GAAG;IACvE,iEAAiE;IACjE,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;IACxB,wEAAwE;IACxE,QAAQ,CAAC,YAAY,EAAE,OAAO,CAAC;IAC/B,qDAAqD;IACrD,QAAQ,CAAC,aAAa,EAAE,MAAM,CAAC;CAChC,CAiBA;AAED,iDAAiD;AACjD,wBAAgB,YAAY,CAAC,YAAY,EAAE,MAAM,GAAG,MAAM,CAuBzD;AAED,OAAO,EAAE,oBAAoB,EAAE,WAAW,EAAE,CAAC;AAC7C,YAAY,EAAE,WAAW,EAAE,iBAAiB,EAAE,SAAS,EAAE,CAAC"}
@@ -0,0 +1,143 @@
1
+ /**
2
+ * A survey over a real host's own session store.
3
+ *
4
+ * ── What this measures, and what it does not ──────────────────────────────
5
+ *
6
+ * It answers ONE question, on real data, with no model in the loop:
7
+ *
8
+ * *given a real session's real token accounting, what would a bounded
9
+ * Aₜ = (P, Σₜ, Oₜ) prompt have cost on those same steps?*
10
+ *
11
+ * It does NOT answer whether an agent given that bounded prompt still does the
12
+ * work. That is an outcome question, it needs a live model, and no amount of
13
+ * offline arithmetic substitutes for it. A survey that reported a saving as if
14
+ * it established the claim would be repeating the original error with extra
15
+ * steps — the original A/B reported a number for a run whose integration had
16
+ * never engaged, and a cost-only win with no task completion is worth nothing.
17
+ *
18
+ * The two halves are complementary: this establishes that the cost side is
19
+ * real and large, the A/B establishes that the work still gets done.
20
+ *
21
+ * ── Why the host's own store ─────────────────────────────────────────────
22
+ *
23
+ * OpenCode records per-message `tokens` — `input`, `cache.read`, `output`,
24
+ * `reasoning`. Reading them means the numbers are the host's accounting for a
25
+ * real session rather than our reconstruction of it, and it works with no live
26
+ * model, which is what makes it usable when a provider quota is exhausted.
27
+ *
28
+ * `cache.read` is the field that carries the finding. A growing transcript is
29
+ * not merely expensive to send: the host caches the prefix, so re-reading
30
+ * history is billed as cache reads. Counting only fresh `input` would report a
31
+ * few percent and miss the entire effect.
32
+ *
33
+ * @non-paper — measurement infrastructure, not the paper's evaluation.
34
+ */
35
+ import { assessReconstruction, reconstruct } from './replay.js';
36
+ /**
37
+ * Survey many sessions at once.
38
+ *
39
+ * Aggregates rather than averaging the per-session percentages: a 1123-step
40
+ * session and a 3-step session must not count equally, and taking a mean of
41
+ * ratios would let the many short cheap sessions drown the few long expensive
42
+ * ones that carry the finding.
43
+ */
44
+ export function survey(sessions, options = {}) {
45
+ const results = [];
46
+ for (const session of sessions) {
47
+ const result = reconstruct(session, {
48
+ ...(options.boundedPromptTokens === undefined
49
+ ? {}
50
+ : { boundedPromptTokens: options.boundedPromptTokens }),
51
+ });
52
+ results.push(options.keepPerStep === true
53
+ ? result
54
+ : { ...result, hostPerStep: [], boundedPerStep: [] });
55
+ }
56
+ const hostPromptTokens = results.reduce((sum, r) => sum + r.hostPromptTokens, 0);
57
+ const boundedPromptTokens = results.reduce((sum, r) => sum + r.boundedPromptTokens, 0);
58
+ const savedTokens = hostPromptTokens - boundedPromptTokens;
59
+ const freshInputTokens = results.reduce((sum, r) => sum + r.freshInputTokens, 0);
60
+ const cacheReadTokens = results.reduce((sum, r) => sum + r.cacheReadTokens, 0);
61
+ const hostEffectiveTokens = results.reduce((sum, r) => sum + r.hostEffectiveTokens, 0);
62
+ const perSession = results
63
+ .map((r) => r.savedFraction)
64
+ .filter((f) => f !== null)
65
+ .sort((a, b) => a - b);
66
+ const middle = perSession.length >> 1;
67
+ const medianSavedFraction = perSession.length === 0
68
+ ? 0
69
+ : perSession.length % 2 === 1
70
+ ? perSession[middle]
71
+ : (perSession[middle - 1] + perSession[middle]) / 2;
72
+ return {
73
+ sessions: results.length,
74
+ steps: results.reduce((sum, r) => sum + r.steps, 0),
75
+ hostPromptTokens,
76
+ boundedPromptTokens,
77
+ savedTokens,
78
+ savedFraction: hostPromptTokens === 0 ? null : savedTokens / hostPromptTokens,
79
+ freshInputTokens,
80
+ cacheReadTokens,
81
+ cacheShare: hostPromptTokens === 0 ? 0 : cacheReadTokens / hostPromptTokens,
82
+ hostEffectiveTokens,
83
+ savedEffectiveFraction: hostEffectiveTokens === 0
84
+ ? null
85
+ : (hostEffectiveTokens - boundedPromptTokens) / hostEffectiveTokens,
86
+ perSession,
87
+ medianSavedFraction,
88
+ grewCount: results.filter((r) => r.hostSlope > 0).length,
89
+ boundedLosesCount: results.filter((r) => r.savedTokens < 0).length,
90
+ spentSessions: results.filter((r) => r.hostPromptTokens > 0).length,
91
+ results: [...results].sort((a, b) => b.savedTokens - a.savedTokens),
92
+ };
93
+ }
94
+ /**
95
+ * How large a bounded prompt may be before it stops being the cheaper option.
96
+ *
97
+ * The honest way to state the finding: not "a bounded prompt saves 99%", but
98
+ * "a bounded prompt of up to N tokens per step is cheaper in EVERY session
99
+ * measured". N is a property of the data, not a number chosen to look good.
100
+ */
101
+ export function breakEvenPromptTokens(sessions) {
102
+ const averages = sessions
103
+ .filter((s) => s.steps.length > 0)
104
+ .map((s) => s.steps.reduce((sum, step) => sum + step.input + step.cacheRead, 0) / s.steps.length)
105
+ .sort((a, b) => a - b);
106
+ if (averages.length === 0) {
107
+ return { tokens: 0, conservative: true, medianAverage: 0 };
108
+ }
109
+ const middle = averages.length >> 1;
110
+ return {
111
+ tokens: Math.floor(averages[0]),
112
+ conservative: true,
113
+ medianAverage: averages.length % 2 === 1
114
+ ? averages[middle]
115
+ : (averages[middle - 1] + averages[middle]) / 2,
116
+ };
117
+ }
118
+ /** Render a survey as a human-readable block. */
119
+ export function formatSurvey(surveyResult) {
120
+ const pct = (value) => value === null ? 'n/a' : `${(value * 100).toFixed(1)}%`;
121
+ return [
122
+ `sessions : ${surveyResult.sessions}`,
123
+ `steps total : ${surveyResult.steps.toLocaleString('en-US')}`,
124
+ '',
125
+ ' raw token count (inflated - counts a cache read as a fresh input):',
126
+ ` fresh input : ${surveyResult.freshInputTokens.toLocaleString('en-US')}`,
127
+ ` cache read : ${surveyResult.cacheReadTokens.toLocaleString('en-US')} (${(surveyResult.cacheShare * 100).toFixed(1)}% of prompts)`,
128
+ ` host total : ${surveyResult.hostPromptTokens.toLocaleString('en-US')}`,
129
+ ` bounded A_t : ${surveyResult.boundedPromptTokens.toLocaleString('en-US')}`,
130
+ ` raw saving : ${pct(surveyResult.savedFraction)}`,
131
+ '',
132
+ ' priced in input-equivalent tokens (cache reads discounted) - QUOTE THIS:',
133
+ ` host effective : ${Math.round(surveyResult.hostEffectiveTokens).toLocaleString('en-US')}`,
134
+ ` bounded A_t : ${surveyResult.boundedPromptTokens.toLocaleString('en-US')}`,
135
+ ` real saving : ${pct(surveyResult.savedEffectiveFraction)}`,
136
+ '',
137
+ `median per session : ${pct(surveyResult.medianSavedFraction)}`,
138
+ `transcript grew : ${surveyResult.grewCount}/${surveyResult.sessions}`,
139
+ `bounded loses : ${surveyResult.boundedLosesCount}/${surveyResult.spentSessions} real runs`,
140
+ ].join('\n');
141
+ }
142
+ export { assessReconstruction, reconstruct };
143
+ //# sourceMappingURL=survey.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"survey.js","sourceRoot":"","sources":["../../src/ab/survey.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAiCG;AAEH,OAAO,EAAE,oBAAoB,EAAE,WAAW,EAAE,MAAM,aAAa,CAAC;AAuEhE;;;;;;;GAOG;AACH,MAAM,UAAU,MAAM,CACpB,QAAgC,EAChC,OAAO,GAAkB,EAAE;IAE3B,MAAM,OAAO,GAAwB,EAAE,CAAC;IACxC,KAAK,MAAM,OAAO,IAAI,QAAQ,EAAE,CAAC;QAC/B,MAAM,MAAM,GAAG,WAAW,CAAC,OAAO,EAAE;YAClC,GAAG,CAAC,OAAO,CAAC,mBAAmB,KAAK,SAAS;gBAC3C,CAAC,CAAC,EAAE;gBACJ,CAAC,CAAC,EAAE,mBAAmB,EAAE,OAAO,CAAC,mBAAmB,EAAE,CAAC;SAC1D,CAAC,CAAC;QACH,OAAO,CAAC,IAAI,CACV,OAAO,CAAC,WAAW,KAAK,IAAI;YAC1B,CAAC,CAAC,MAAM;YACR,CAAC,CAAC,EAAE,GAAG,MAAM,EAAE,WAAW,EAAE,EAAE,EAAE,cAAc,EAAE,EAAE,EAAE,CACvD,CAAC;IACJ,CAAC;IAED,MAAM,gBAAgB,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,CAAC,gBAAgB,EAAE,CAAC,CAAC,CAAC;IACjF,MAAM,mBAAmB,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,CAAC,mBAAmB,EAAE,CAAC,CAAC,CAAC;IACvF,MAAM,WAAW,GAAG,gBAAgB,GAAG,mBAAmB,CAAC;IAC3D,MAAM,gBAAgB,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,CAAC,gBAAgB,EAAE,CAAC,CAAC,CAAC;IACjF,MAAM,eAAe,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,CAAC,eAAe,EAAE,CAAC,CAAC,CAAC;IAC/E,MAAM,mBAAmB,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,CAAC,mBAAmB,EAAE,CAAC,CAAC,CAAC;IACvF,MAAM,UAAU,GAAG,OAAO;SACvB,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,aAAa,CAAC;SAC3B,MAAM,CAAC,CAAC,CAAC,EAAe,EAAE,CAAC,CAAC,KAAK,IAAI,CAAC;SACtC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IACzB,MAAM,MAAM,GAAG,UAAU,CAAC,MAAM,IAAI,CAAC,CAAC;IACtC,MAAM,mBAAmB,GACvB,UAAU,CAAC,MAAM,KAAK,CAAC;QACrB,CAAC,CAAC,CAAC;QACH,CAAC,CAAC,UAAU,CAAC,MAAM,GAAG,CAAC,KAAK,CAAC;YAC3B,CAAC,CAAC,UAAU,CAAC,MAAM,CAAE;YACrB,CAAC,CAAC,CAAC,UAAU,CAAC,MAAM,GAAG,CAAC,CAAE,GAAG,UAAU,CAAC,MAAM,CAAE,CAAC,GAAG,CAAC,CAAC;IAE5D,OAAO;QACL,QAAQ,EAAE,OAAO,CAAC,MAAM;QACxB,KAAK,EAAE,OAAO,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,CAAC,KAAK,EAAE,CAAC,CAAC;QACnD,gBAAgB;QAChB,mBAAmB;QACnB,WAAW;QACX,aAAa,EAAE,gBAAgB,KAAK,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,WAAW,GAAG,gBAAgB;QAC7E,gBAAgB;QAChB,eAAe;QACf,UAAU,EAAE,gBAAgB,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,eAAe,GAAG,gBAAgB;QAC3E,mBAAmB;QACnB,sBAAsB,EACpB,mBAAmB,KAAK,CAAC;YACvB,CAAC,CAAC,IAAI;YACN,CAAC,CAAC,CAAC,mBAAmB,GAAG,mBAAmB,CAAC,GAAG,mBAAmB;QACvE,UAAU;QACV,mBAAmB;QACnB,SAAS,EAAE,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,SAAS,GAAG,CAAC,CAAC,CAAC,MAAM;QACxD,iBAAiB,EAAE,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,WAAW,GAAG,CAAC,CAAC,CAAC,MAAM;QAClE,aAAa,EAAE,OAAO,CAAC,MAAM,CAC3B,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,gBAAgB,GAAG,CAAC,CAC9B,CAAC,MAAM;QACR,OAAO,EAAE,CAAC,GAAG,OAAO,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,WAAW,GAAG,CAAC,CAAC,WAAW,CAAC;KACpE,CAAC;AACJ,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,qBAAqB,CAAC,QAAgC;IAQpE,MAAM,QAAQ,GAAG,QAAQ;SACtB,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC;SACjC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,IAAI,EAAE,EAAE,CAAC,GAAG,GAAG,IAAI,CAAC,KAAK,GAAG,IAAI,CAAC,SAAS,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,KAAK,CAAC,MAAM,CAAC;SAChG,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IACzB,IAAI,QAAQ,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QAC1B,OAAO,EAAE,MAAM,EAAE,CAAC,EAAE,YAAY,EAAE,IAAI,EAAE,aAAa,EAAE,CAAC,EAAE,CAAC;IAC7D,CAAC;IACD,MAAM,MAAM,GAAG,QAAQ,CAAC,MAAM,IAAI,CAAC,CAAC;IACpC,OAAO;QACL,MAAM,EAAE,IAAI,CAAC,KAAK,CAAC,QAAQ,CAAC,CAAC,CAAE,CAAC;QAChC,YAAY,EAAE,IAAI;QAClB,aAAa,EACX,QAAQ,CAAC,MAAM,GAAG,CAAC,KAAK,CAAC;YACvB,CAAC,CAAC,QAAQ,CAAC,MAAM,CAAE;YACnB,CAAC,CAAC,CAAC,QAAQ,CAAC,MAAM,GAAG,CAAC,CAAE,GAAG,QAAQ,CAAC,MAAM,CAAE,CAAC,GAAG,CAAC;KACtD,CAAC;AACJ,CAAC;AAED,iDAAiD;AACjD,MAAM,UAAU,YAAY,CAAC,YAAoB;IAC/C,MAAM,GAAG,GAAG,CAAC,KAAoB,EAAU,EAAE,CAC3C,KAAK,KAAK,IAAI,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,GAAG,CAAC,KAAK,GAAG,GAAG,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,GAAG,CAAC;IAC1D,OAAO;QACL,yBAAyB,YAAY,CAAC,QAAQ,EAAE;QAChD,yBAAyB,YAAY,CAAC,KAAK,CAAC,cAAc,CAAC,OAAO,CAAC,EAAE;QACrE,EAAE;QACF,sEAAsE;QACtE,0BAA0B,YAAY,CAAC,gBAAgB,CAAC,cAAc,CAAC,OAAO,CAAC,EAAE;QACjF,0BAA0B,YAAY,CAAC,eAAe,CAAC,cAAc,CAAC,OAAO,CAAC,KAAK,CAAC,YAAY,CAAC,UAAU,GAAG,GAAG,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,eAAe;QAC5I,0BAA0B,YAAY,CAAC,gBAAgB,CAAC,cAAc,CAAC,OAAO,CAAC,EAAE;QACjF,0BAA0B,YAAY,CAAC,mBAAmB,CAAC,cAAc,CAAC,OAAO,CAAC,EAAE;QACpF,0BAA0B,GAAG,CAAC,YAAY,CAAC,aAAa,CAAC,EAAE;QAC3D,EAAE;QACF,4EAA4E;QAC5E,0BAA0B,IAAI,CAAC,KAAK,CAAC,YAAY,CAAC,mBAAmB,CAAC,CAAC,cAAc,CAAC,OAAO,CAAC,EAAE;QAChG,0BAA0B,YAAY,CAAC,mBAAmB,CAAC,cAAc,CAAC,OAAO,CAAC,EAAE;QACpF,0BAA0B,GAAG,CAAC,YAAY,CAAC,sBAAsB,CAAC,EAAE;QACpE,EAAE;QACF,yBAAyB,GAAG,CAAC,YAAY,CAAC,mBAAmB,CAAC,EAAE;QAChE,yBAAyB,YAAY,CAAC,SAAS,IAAI,YAAY,CAAC,QAAQ,EAAE;QAC1E,yBAAyB,YAAY,CAAC,iBAAiB,IAAI,YAAY,CAAC,aAAa,YAAY;KAClG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AACf,CAAC;AAED,OAAO,EAAE,oBAAoB,EAAE,WAAW,EAAE,CAAC"}
@@ -0,0 +1,99 @@
1
+ /**
2
+ * Token accounting read from the HOST's own store.
3
+ *
4
+ * ── Why not parse the host's stdout ──────────────────────────────────────
5
+ *
6
+ * Three reasons, in order of how much they matter:
7
+ *
8
+ * 1. **The model does not report its own cost reliably.** A harness that
9
+ * takes the number the subject of the experiment wrote down is trusting the
10
+ * measurement to the thing being measured. The previous A/B took its
11
+ * figures from the run's own reporting, and one of its two "arms" was
12
+ * itself two different readings of a single session.
13
+ * 2. **OpenCode already records per-message token usage** in its local store
14
+ * (`message.data.tokens` with `input`, `output`, `reasoning` and
15
+ * `cache.{read,write}`). That is the host's own accounting, not a
16
+ * reconstruction, and it is keyed by session so a run can be re-read
17
+ * after the fact.
18
+ * 3. **It is available without a live model.** Reading the store works when
19
+ * the provider quota is exhausted, which is exactly when one most wants to
20
+ * re-analyse a previous run.
21
+ *
22
+ * Node's `node:sqlite` is used rather than a dependency: the store is a
23
+ * local file, the read is a single prepared statement, and adding a native
24
+ * dependency to a zero-deps package for this would be a poor trade. The
25
+ * reader is injected, so the query and its failure modes are testable
26
+ * without an OpenCode installation.
27
+ *
28
+ * @non-paper — measurement infrastructure, not the paper's evaluation.
29
+ */
30
+ import type { TokenUsage } from './record.js';
31
+ /** One row of the host's message table, as far as this module cares. */
32
+ export interface HostMessageRow {
33
+ /** The session the message belongs to. */
34
+ readonly sessionID: string;
35
+ /** Milliseconds since the epoch. */
36
+ readonly created: number;
37
+ /** `assistant`, `user`, or anything else this module ignores. */
38
+ readonly role: string;
39
+ /** Token accounting as the host recorded it. */
40
+ readonly tokens: Partial<TokenUsage> | null;
41
+ }
42
+ /**
43
+ * The narrow seam onto the host's store.
44
+ *
45
+ * Injecting this is what lets the arithmetic below be tested against
46
+ * hand-written rows instead of against a live OpenCode server, and what lets
47
+ * a caller substitute a different source (a JSON export, a CI fixture)
48
+ * without touching the logic that consumes it.
49
+ */
50
+ export interface UsageReader {
51
+ /**
52
+ * Every message recorded for `sessionID`, in creation order.
53
+ *
54
+ * Rejects when the store is missing, locked, or the session is unknown —
55
+ * an unreadable store must surface as a failed run, never as zero tokens,
56
+ * because zero tokens would make an arm look free.
57
+ */
58
+ messagesFor(sessionID: string): Promise<readonly HostMessageRow[]>;
59
+ }
60
+ /** Zero usage, for a session whose messages carry no accounting. */
61
+ export declare const NO_USAGE: TokenUsage;
62
+ /**
63
+ * Total token spend for a session.
64
+ *
65
+ * Only `assistant` rows count. A user message's `input` figure is the host
66
+ * describing the prompt it echoed, and adding it would double-count the
67
+ * conversation.
68
+ *
69
+ * `cacheRead` is summed, not treated as free: prompt caching changes the
70
+ * price of a token, not the fact that the model was shown it, and the
71
+ * paper's claim is about what the model is exposed to.
72
+ */
73
+ export declare function sessionUsage(rows: readonly HostMessageRow[]): TokenUsage;
74
+ /** Why a usage lookup produced nothing. */
75
+ export type UsageFailure =
76
+ /** The host store could not be opened or read. */
77
+ 'store_unavailable'
78
+ /** The session id is not in the host's store. */
79
+ | 'session_unknown';
80
+ /** The outcome of resolving one session's token spend. */
81
+ export type UsageOutcome = {
82
+ readonly ok: true;
83
+ readonly usage: TokenUsage;
84
+ readonly assistantMessages: number;
85
+ } | {
86
+ readonly ok: false;
87
+ readonly reason: UsageFailure;
88
+ readonly detail: string;
89
+ };
90
+ /**
91
+ * Resolve a session's token spend, turning every failure into a value.
92
+ *
93
+ * A run whose accounting cannot be read is a run with unknown cost, which is
94
+ * not the same as a run that cost nothing. The distinction is the whole
95
+ * reason this returns a union instead of a number: `store_unavailable` must
96
+ * never be silently read as `NO_USAGE`.
97
+ */
98
+ export declare function resolveSessionUsage(reader: UsageReader, sessionID: string): Promise<UsageOutcome>;
99
+ //# sourceMappingURL=usage.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"usage.d.ts","sourceRoot":"","sources":["../../src/ab/usage.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA4BG;AAEH,OAAO,KAAK,EAAE,UAAU,EAAE,MAAM,aAAa,CAAC;AAE9C,wEAAwE;AACxE,MAAM,WAAW,cAAc;IAC7B,0CAA0C;IAC1C,QAAQ,CAAC,SAAS,EAAE,MAAM,CAAC;IAC3B,oCAAoC;IACpC,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAC;IACzB,iEAAiE;IACjE,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAC;IACtB,gDAAgD;IAChD,QAAQ,CAAC,MAAM,EAAE,OAAO,CAAC,UAAU,CAAC,GAAG,IAAI,CAAC;CAC7C;AAED;;;;;;;GAOG;AACH,MAAM,WAAW,WAAW;IAC1B;;;;;;OAMG;IACH,WAAW,CAAC,SAAS,EAAE,MAAM,GAAG,OAAO,CAAC,SAAS,cAAc,EAAE,CAAC,CAAC;CACpE;AAED,oEAAoE;AACpE,eAAO,MAAM,QAAQ,EAAE,UAKtB,CAAC;AAYF;;;;;;;;;;GAUG;AACH,wBAAgB,YAAY,CAAC,IAAI,EAAE,SAAS,cAAc,EAAE,GAAG,UAAU,CAaxE;AAED,2CAA2C;AAC3C,MAAM,MAAM,YAAY;AACtB,kDAAkD;AAChD,mBAAmB;AACrB,iDAAiD;GAC/C,iBAAiB,CAAC;AAEtB,0DAA0D;AAC1D,MAAM,MAAM,YAAY,GACpB;IAAE,QAAQ,CAAC,EAAE,EAAE,IAAI,CAAC;IAAC,QAAQ,CAAC,KAAK,EAAE,UAAU,CAAC;IAAC,QAAQ,CAAC,iBAAiB,EAAE,MAAM,CAAA;CAAE,GACrF;IAAE,QAAQ,CAAC,EAAE,EAAE,KAAK,CAAC;IAAC,QAAQ,CAAC,MAAM,EAAE,YAAY,CAAC;IAAC,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAA;CAAE,CAAC;AAEnF;;;;;;;GAOG;AACH,wBAAsB,mBAAmB,CACvC,MAAM,EAAE,WAAW,EACnB,SAAS,EAAE,MAAM,GAChB,OAAO,CAAC,YAAY,CAAC,CAoBvB"}