@tangle-network/agent-eval 0.173.2 → 0.174.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -0
- package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
- package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
- package/dist/adapters/http.d.ts +2 -2
- package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
- package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +7 -9
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
- package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
- package/dist/{benchmark-command-CS6gVHVq.js → benchmark-command-mZIlR-ra.js} +13 -13
- package/dist/{benchmark-command-CS6gVHVq.js.map → benchmark-command-mZIlR-ra.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +3 -4
- package/dist/benchmarks/index.d.ts.map +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +5 -9
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
- package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
- package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -10
- package/dist/contract/index.js +8 -8
- package/dist/{default-registry-B0bKikCb.js → default-registry-CrAp0pYq.js} +4 -4
- package/dist/{default-registry-B0bKikCb.js.map → default-registry-CrAp0pYq.js.map} +1 -1
- package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
- package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
- package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
- package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
- package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
- package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
- package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
- package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
- package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +1 -4
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
- package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/fuzz.js.map +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
- package/dist/index-BTrx5s8m.d.ts.map +1 -0
- package/dist/{index-D0Db5X-4.d.ts → index-Bn-nlnSV.d.ts} +4 -4
- package/dist/{index-D0Db5X-4.d.ts.map → index-Bn-nlnSV.d.ts.map} +1 -1
- package/dist/index-DKXuBPXf.d.ts +3840 -0
- package/dist/index-DKXuBPXf.d.ts.map +1 -0
- package/dist/index.d.ts +11 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +10 -10
- package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
- package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
- package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
- package/dist/llm-judge-DmNaBrXB.js.map +1 -0
- package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
- package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
- package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
- package/dist/rl.d.ts +1 -1
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js.map +1 -1
- package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
- package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
- package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
- package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
- package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
- package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
- package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
- package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
- package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
- package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
- package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +25 -7
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
- package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
- package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
- package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/traces.js +4 -4
- package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
- package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
- package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
- package/dist/types-DQ0e2E7y.js.map +1 -0
- package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
- package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
- package/docs/campaign-proposers.md +42 -0
- package/package.json +1 -1
- package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
- package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
- package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
- package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
- package/dist/benchmark-BjLGkfnN.d.ts +0 -236
- package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
- package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
- package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
- package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
- package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
- package/dist/index-BQqOjerE.d.ts.map +0 -1
- package/dist/index-CFDffsKz.d.ts +0 -1135
- package/dist/index-CFDffsKz.d.ts.map +0 -1
- package/dist/llm-judge-BfqMFo4h.js.map +0 -1
- package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
- package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
- package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
- package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
- package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
- package/dist/proposal-findings-bko3GGy-.js.map +0 -1
- package/dist/provenance-CRY67X50.d.ts +0 -1995
- package/dist/provenance-CRY67X50.d.ts.map +0 -1
- package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
- package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
- package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
- package/dist/types-CiWITkGo.js.map +0 -1
|
@@ -0,0 +1,1127 @@
|
|
|
1
|
+
import { a as RunRecord } from "./run-record-DTv1MdjK.js";
|
|
2
|
+
import { a as PairedPromotionDecision, d as PairedBootstrapResult, i as PairedMcNemarEvidence, r as PairedDecisionStatistic, t as PairedDecisionMethod, u as PairedBootstrapOptions } from "./paired-promotion-decision-CGzg0cI_.js";
|
|
3
|
+
import { R as Scenario, b as GenerationRecord, h as GateContext, p as Gate, v as GateResult, w as JudgeScore } from "./types-BJz2CPTM.js";
|
|
4
|
+
//#region src/statistics/paired-binary.d.ts
|
|
5
|
+
/** A binomial proportion estimate with a confidence interval. */
|
|
6
|
+
interface ProportionInterval {
|
|
7
|
+
/** Point estimate successes / n (0 when n = 0). */
|
|
8
|
+
estimate: number;
|
|
9
|
+
/** Lower bound, clamped to [0, 1]. */
|
|
10
|
+
lower: number;
|
|
11
|
+
/** Upper bound, clamped to [0, 1]. */
|
|
12
|
+
upper: number;
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* Wilson score interval for a binomial proportion. Correct at small n and near
|
|
16
|
+
* 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
|
|
17
|
+
* understates coverage. Use this for any pass-rate / hit-rate / realness-rate
|
|
18
|
+
* CI — the continuous `confidenceInterval` assumes the wrong distribution for a
|
|
19
|
+
* proportion. `n = 0 ⇒ {0, 0, 0}`.
|
|
20
|
+
*/
|
|
21
|
+
declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
|
|
22
|
+
/**
|
|
23
|
+
* Are these per-item outcomes binary (every value exactly 0 or 1)?
|
|
24
|
+
*
|
|
25
|
+
* The discriminator a promotion gate needs before choosing a paired statistic.
|
|
26
|
+
* On binary outcomes the paired delta vector lives in {-1, 0, +1} and is
|
|
27
|
+
* normally dominated by zeros (both arms solve, or both arms miss, most items),
|
|
28
|
+
* so its MEDIAN is pinned at exactly 0 no matter how large the real shift in
|
|
29
|
+
* success rate is — and a bootstrap CI on that median collapses to [0, 0].
|
|
30
|
+
* A gate keying on `ci.low > threshold` is then structurally unable to see
|
|
31
|
+
* either a gain or a regression. Detect this shape and switch to the
|
|
32
|
+
* paired-binary estimators ({@link mcnemar}, {@link pairedRiskDifference})
|
|
33
|
+
* instead of silently answering "no" forever.
|
|
34
|
+
*
|
|
35
|
+
* Empty input is NOT binary: there is no evidence of the outcome's shape, and
|
|
36
|
+
* defaulting an empty vector into the binary branch would pick a statistic on
|
|
37
|
+
* no data at all.
|
|
38
|
+
*
|
|
39
|
+
* NOT the right discriminator for a gate. It recognises the literal {0, 1}
|
|
40
|
+
* encoding and nothing else, so a pass/fail dimension emitted on 0-100 — which
|
|
41
|
+
* judges in this codebase do routinely — reads as non-binary, and a single
|
|
42
|
+
* partial-credit score in an otherwise pass/fail vector flips it to false while
|
|
43
|
+
* leaving the median just as blind. Gates want {@link pairedBinaryScale} (any
|
|
44
|
+
* two-point encoding). This predicate remains for callers that specifically
|
|
45
|
+
* mean "literally 0/1".
|
|
46
|
+
*/
|
|
47
|
+
declare function isBinaryOutcomeVector(values: ArrayLike<number>): boolean;
|
|
48
|
+
/** Result of a McNemar paired-binary significance test. */
|
|
49
|
+
interface McNemarResult {
|
|
50
|
+
/** Total paired observations. */
|
|
51
|
+
n: number;
|
|
52
|
+
/** Discordant pairs (b + c) — the only ones that carry signal. */
|
|
53
|
+
nDiscordant: number;
|
|
54
|
+
/** Pairs where treatment succeeded and control failed ("newly correct"). */
|
|
55
|
+
b: number;
|
|
56
|
+
/** Pairs where control succeeded and treatment failed ("newly wrong"). */
|
|
57
|
+
c: number;
|
|
58
|
+
/** Continuity-corrected chi-square statistic (reference; exact p drives the call). */
|
|
59
|
+
statistic: number;
|
|
60
|
+
/** Two-sided p-value. Exact (binomial sign test on discordant pairs). */
|
|
61
|
+
pValue: number;
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* McNemar's test for paired binary outcomes — the correct significance test for
|
|
65
|
+
* "does treatment change the success rate vs control on the SAME items". Only
|
|
66
|
+
* discordant pairs (one arm right, the other wrong) carry information; concordant
|
|
67
|
+
* pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
|
|
68
|
+
* rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
|
|
69
|
+
* Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
|
|
70
|
+
* at the small discordant counts typical of eval runs (no continuity-corrected
|
|
71
|
+
* chi-square approximation needed, though it is returned as `statistic` for
|
|
72
|
+
* reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
|
|
73
|
+
* the module's (before, after) convention. Throws on unequal lengths.
|
|
74
|
+
*/
|
|
75
|
+
declare function mcnemar(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>): McNemarResult;
|
|
76
|
+
/** A paired binary effect size (treatment rate − control rate) with a CI. */
|
|
77
|
+
interface RiskDifferenceResult {
|
|
78
|
+
/** Total paired observations. */
|
|
79
|
+
n: number;
|
|
80
|
+
/** Discordant pairs: treatment-win count. */
|
|
81
|
+
b: number;
|
|
82
|
+
/** Discordant pairs: control-win count. */
|
|
83
|
+
c: number;
|
|
84
|
+
/** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
|
|
85
|
+
riskDifference: number;
|
|
86
|
+
/** Lower bound of the CI, clamped to [-1, 1]. */
|
|
87
|
+
lower: number;
|
|
88
|
+
/** Upper bound of the CI, clamped to [-1, 1]. */
|
|
89
|
+
upper: number;
|
|
90
|
+
/** Confidence level used. */
|
|
91
|
+
confidence: number;
|
|
92
|
+
}
|
|
93
|
+
/**
|
|
94
|
+
* Paired risk difference (the effect-size companion to {@link mcnemar}): the
|
|
95
|
+
* change in success rate p(treatment) − p(control) on matched items, which for
|
|
96
|
+
* paired binary data equals (b − c) / n. The CI uses the paired variance from
|
|
97
|
+
* the discordant counts, not the independent-samples formula (which overstates
|
|
98
|
+
* the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
|
|
99
|
+
* arrays, control first. Throws on unequal lengths.
|
|
100
|
+
*
|
|
101
|
+
* REPORTING ONLY — do NOT decide a promotion on this interval. The CI is a Wald
|
|
102
|
+
* normal approximation, which badly UNDERCOVERS when only a handful of pairs are
|
|
103
|
+
* discordant: at n = 3 with b = 2, c = 0 it returns [0.133, 1.000], excluding 0,
|
|
104
|
+
* while McNemar's exact test on the same data gives p = 0.50. A gate keying on
|
|
105
|
+
* `lower > 0` would promote noise. Use {@link pairedRiskDifferenceExact}, whose
|
|
106
|
+
* interval is dual to the exact test by construction, for any decision.
|
|
107
|
+
*/
|
|
108
|
+
declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
|
|
109
|
+
/** A paired binary effect size with an EXACT interval and the exact test that
|
|
110
|
+
* bounds it — one object so a caller cannot read the estimate without the
|
|
111
|
+
* significance it is entitled to. */
|
|
112
|
+
interface ExactRiskDifferenceResult {
|
|
113
|
+
/** Total paired observations. */
|
|
114
|
+
n: number;
|
|
115
|
+
/** Discordant pairs: treatment-win count. */
|
|
116
|
+
b: number;
|
|
117
|
+
/** Discordant pairs: control-win count. */
|
|
118
|
+
c: number;
|
|
119
|
+
/** Discordant pairs (b + c) — the only ones carrying information. */
|
|
120
|
+
nDiscordant: number;
|
|
121
|
+
/** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
|
|
122
|
+
riskDifference: number;
|
|
123
|
+
/** Exact conditional CI lower bound. 0 when there are no discordant pairs. */
|
|
124
|
+
lower: number;
|
|
125
|
+
/** Exact conditional CI upper bound. 0 when there are no discordant pairs. */
|
|
126
|
+
upper: number;
|
|
127
|
+
/** Confidence level used. */
|
|
128
|
+
confidence: number;
|
|
129
|
+
/** McNemar's exact two-sided p-value on the same discordant counts. */
|
|
130
|
+
pValue: number;
|
|
131
|
+
}
|
|
132
|
+
/**
|
|
133
|
+
* Paired risk difference with the EXACT CONDITIONAL interval — the estimator a
|
|
134
|
+
* promotion gate may decide on.
|
|
135
|
+
*
|
|
136
|
+
* Conditional on the number of discordant pairs m = b + c, the treatment-win
|
|
137
|
+
* count b is Binomial(m, π) with π = P(treatment wins | discordant), and the
|
|
138
|
+
* risk difference is an exact reparameterisation: RD = (2π − 1)·m/n. So a
|
|
139
|
+
* Clopper-Pearson exact interval for π maps straight onto RD. This buys the
|
|
140
|
+
* property the Wald interval in {@link pairedRiskDifference} does not have:
|
|
141
|
+
*
|
|
142
|
+
* **`lower > 0` ⟺ McNemar's exact test rejects at α = 1 − confidence.**
|
|
143
|
+
*
|
|
144
|
+
* Clopper-Pearson excludes π = 0.5 exactly when the two-sided exact binomial
|
|
145
|
+
* test of π = 0.5 rejects, and that test IS {@link mcnemar}'s p-value — so the
|
|
146
|
+
* interval and the test can never disagree, and a gate keyed on `lower` cannot
|
|
147
|
+
* promote what the exact test refuses. The exact p is returned in the same
|
|
148
|
+
* object so the two are impossible to compute apart.
|
|
149
|
+
*
|
|
150
|
+
* The interval is conservative (exact intervals over-cover; conditioning on m
|
|
151
|
+
* discards the concordant pairs' information about m itself). That is the
|
|
152
|
+
* correct direction for a promotion gate: it refuses more often, never less.
|
|
153
|
+
*
|
|
154
|
+
* With m = 0 there are no discordant pairs and π is not identified: the result
|
|
155
|
+
* is the degenerate [0, 0] with p = 1. That is NOT evidence of equivalence —
|
|
156
|
+
* callers must treat a zero-width interval as "cannot decide", not as "no
|
|
157
|
+
* difference". Inputs are paired 0/1 (or boolean) arrays, control first.
|
|
158
|
+
* Throws on unequal lengths.
|
|
159
|
+
*/
|
|
160
|
+
declare function pairedRiskDifferenceExact(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ExactRiskDifferenceResult;
|
|
161
|
+
/** A paired binary effect size with an interval that is valid at a NONZERO
|
|
162
|
+
* margin — the estimator a noninferiority decision may be made on. */
|
|
163
|
+
interface ScoreRiskDifferenceResult {
|
|
164
|
+
/** Total paired observations. */
|
|
165
|
+
n: number;
|
|
166
|
+
/** Discordant pairs: treatment-win count. */
|
|
167
|
+
b: number;
|
|
168
|
+
/** Discordant pairs: control-win count. */
|
|
169
|
+
c: number;
|
|
170
|
+
/** Discordant pairs (b + c). */
|
|
171
|
+
nDiscordant: number;
|
|
172
|
+
/** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
|
|
173
|
+
riskDifference: number;
|
|
174
|
+
/** Score-interval lower bound on the population risk difference. */
|
|
175
|
+
lower: number;
|
|
176
|
+
/** Score-interval upper bound on the population risk difference. */
|
|
177
|
+
upper: number;
|
|
178
|
+
/** Confidence level used. */
|
|
179
|
+
confidence: number;
|
|
180
|
+
}
|
|
181
|
+
/**
|
|
182
|
+
* Paired risk difference with TANGO'S (1998) SCORE INTERVAL — the estimator a
|
|
183
|
+
* promotion gate may decide on **at a nonzero margin**.
|
|
184
|
+
*
|
|
185
|
+
* {@link pairedRiskDifferenceExact} conditions on the observed discordant count
|
|
186
|
+
* `m = b + c`, builds a Clopper-Pearson interval for the win share among those
|
|
187
|
+
* `m` pairs, and multiplies by the observed `m/n`. That is exact for testing
|
|
188
|
+
* RD = 0 — it is dual to McNemar — but it is NOT a confidence interval for the
|
|
189
|
+
* population risk difference at a nonzero margin, because the sampling
|
|
190
|
+
* variability of `m/n` itself is discarded. The gap is not academic: with the
|
|
191
|
+
* production caller's `pairedDeltaThreshold: -0.05`, a process whose true risk
|
|
192
|
+
* difference sits exactly on that margin clears a nominal-95 % `lower > margin`
|
|
193
|
+
* check 24.75 % of the time at n = 40 and 43.95 % at n = 76 (2000 replicates
|
|
194
|
+
* each) when the conditional interval decides.
|
|
195
|
+
*
|
|
196
|
+
* Tango's interval inverts the score test of RD = delta, which estimates the
|
|
197
|
+
* nuisance loss rate under each hypothesised delta instead of fixing it at the
|
|
198
|
+
* observed value, so `m` contributes its own uncertainty. It is the method
|
|
199
|
+
* `ratesci::scorepairci` uses for paired risk-difference noninferiority, and it
|
|
200
|
+
* is not conditional, so it stays valid as the margin moves away from zero.
|
|
201
|
+
*
|
|
202
|
+
* The bounds are found by bisecting `tangoScore(delta) = ±z` — the score is
|
|
203
|
+
* monotone decreasing in delta, so each crossing is unique. Inputs are paired
|
|
204
|
+
* 0/1 (or boolean) arrays, control first. Throws on unequal lengths.
|
|
205
|
+
*/
|
|
206
|
+
declare function pairedRiskDifferenceScore(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ScoreRiskDifferenceResult;
|
|
207
|
+
/**
|
|
208
|
+
* The common positive level `s` such that EVERY value across both paired arms is
|
|
209
|
+
* exactly 0 or `s` — i.e. the outcome is pass/fail, whatever encoding it arrived
|
|
210
|
+
* in. Returns null when the outcomes are not two-point, when the two arms use
|
|
211
|
+
* different levels, or when no positive value was observed at all (all-zero
|
|
212
|
+
* arms: the level is not identified, and there is nothing to decide anyway).
|
|
213
|
+
*
|
|
214
|
+
* This is the scale-aware successor to {@link isBinaryOutcomeVector}, which only
|
|
215
|
+
* recognises literal {0, 1}. Judges in this codebase emit dimensions on 0-100 as
|
|
216
|
+
* well as [0,1] (see `detectScale` in `campaign/gates/statistical-heldout.ts`),
|
|
217
|
+
* so a pass/fail dimension routinely arrives as {0, 100} and a {0,1}-only test
|
|
218
|
+
* silently sends it down the median path that cannot see it. Any positive level
|
|
219
|
+
* is accepted, not just 1 and 100: for a two-point {0, s} outcome the mean paired
|
|
220
|
+
* delta is exactly s·(b − c)/n, so the binary estimators apply after dividing by
|
|
221
|
+
* s and rescaling the result back into the caller's native units.
|
|
222
|
+
*
|
|
223
|
+
* Non-finite values ⇒ null: an unusable outcome must not be classified as a
|
|
224
|
+
* clean pass/fail shape.
|
|
225
|
+
*/
|
|
226
|
+
declare function pairedBinaryScale(before: ArrayLike<number>, after: ArrayLike<number>): number | null;
|
|
227
|
+
/**
|
|
228
|
+
* Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
|
|
229
|
+
* Language Models Trained on Code"). Given `n` independent samples for one
|
|
230
|
+
* problem of which `c` pass, the probability that at least one of a random k of
|
|
231
|
+
* them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
|
|
232
|
+
* first k pass" is biased high at small n; this is the variance-reduced estimator
|
|
233
|
+
* averaged implicitly over all k-subsets. Average the per-problem values across
|
|
234
|
+
* the suite for the corpus pass@k. Computed in the numerically stable product
|
|
235
|
+
* form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
|
|
236
|
+
*/
|
|
237
|
+
declare function passAtK(n: number, c: number, k: number): number;
|
|
238
|
+
//#endregion
|
|
239
|
+
//#region src/statistics/sequential-eprocess.d.ts
|
|
240
|
+
interface EProcessOptions {
|
|
241
|
+
/** Type-I error budget. The process decides when wealth ≥ 1/alpha
|
|
242
|
+
* (Ville's inequality). Default 0.05. */
|
|
243
|
+
alpha?: number;
|
|
244
|
+
/** Truncation bound on the predictable bet λ ∈ [0, maxBet]. Must satisfy
|
|
245
|
+
* maxBet < 1/nullMean so every wealth factor stays strictly positive.
|
|
246
|
+
* Default 0.5. */
|
|
247
|
+
maxBet?: number;
|
|
248
|
+
/** The null boundary m₀ for H0: E[x] ≤ m₀ on x ∈ [0,1]. Default 0.5
|
|
249
|
+
* (the paired-delta encoding x = (d+1)/2 maps "no effect" to 1/2).
|
|
250
|
+
* A pre-registered minEffect shifts this — see `sequentialPairedGate`. */
|
|
251
|
+
nullMean?: number;
|
|
252
|
+
/** Continue a process from a `state()` snapshot taken before a restart.
|
|
253
|
+
* The snapshot never supplies the parameters: `alpha`, `maxBet`, and
|
|
254
|
+
* `nullMean` resolve from this options object exactly as for a fresh
|
|
255
|
+
* process, and a snapshot recorded under different parameters is refused
|
|
256
|
+
* (`ValidationError`) — re-deciding the same stream under new parameters
|
|
257
|
+
* would reopen optional stopping. A snapshot whose fields are not
|
|
258
|
+
* mutually consistent (tampered or truncated) is refused the same way. */
|
|
259
|
+
resume?: EProcessState;
|
|
260
|
+
}
|
|
261
|
+
interface EProcessStep {
|
|
262
|
+
/** Current wealth W_n — the e-value against H0 after n observations. */
|
|
263
|
+
wealth: number;
|
|
264
|
+
/** Observations consumed so far. */
|
|
265
|
+
n: number;
|
|
266
|
+
/** True from the first n where W_n ≥ 1/alpha onward (sticky). */
|
|
267
|
+
decided: boolean;
|
|
268
|
+
}
|
|
269
|
+
/**
|
|
270
|
+
* Complete snapshot of an e-process. Together with the parameters it is
|
|
271
|
+
* sufficient to continue the process after a restart: `sumX` and `varSum`
|
|
272
|
+
* are the running sums the next bet is computed from, so a process rebuilt
|
|
273
|
+
* from a snapshot produces the same wealth sequence and decision as one that
|
|
274
|
+
* was never interrupted. Plain data: a JSON round-trip preserves it
|
|
275
|
+
* (`decidedAtN` is undefined, and therefore omitted, until decided).
|
|
276
|
+
*/
|
|
277
|
+
interface EProcessState extends EProcessStep {
|
|
278
|
+
alpha: number;
|
|
279
|
+
maxBet: number;
|
|
280
|
+
nullMean: number;
|
|
281
|
+
/** The decision boundary 1/alpha. */
|
|
282
|
+
threshold: number;
|
|
283
|
+
/** Observation count at the first threshold crossing; undefined until decided. */
|
|
284
|
+
decidedAtN?: number;
|
|
285
|
+
/** Σ x_i over the n observations consumed. */
|
|
286
|
+
sumX: number;
|
|
287
|
+
/** Σ (x_i − μ̂_i)² over the n observations consumed, μ̂_i the shrunk running
|
|
288
|
+
* mean after observation i. */
|
|
289
|
+
varSum: number;
|
|
290
|
+
}
|
|
291
|
+
interface EProcess {
|
|
292
|
+
/** Consume one observation x ∈ [0,1]. Throws on non-finite / out-of-range
|
|
293
|
+
* input — a silent clamp would corrupt the type-I guarantee. */
|
|
294
|
+
update(x: number): EProcessStep;
|
|
295
|
+
state(): EProcessState;
|
|
296
|
+
}
|
|
297
|
+
/**
|
|
298
|
+
* Betting test-martingale for bounded observations — the e-process core of
|
|
299
|
+
* anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
|
|
300
|
+
* of bounded random variables by betting", JRSS-B 2024).
|
|
301
|
+
*
|
|
302
|
+
* Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
|
|
303
|
+
*
|
|
304
|
+
* W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
|
|
305
|
+
*
|
|
306
|
+
* with the truncated GROW-style plug-in bet computed from PRIOR observations:
|
|
307
|
+
*
|
|
308
|
+
* λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
|
|
309
|
+
*
|
|
310
|
+
* where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
|
|
311
|
+
* σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
|
|
312
|
+
*
|
|
313
|
+
* PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
|
|
314
|
+
* ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
|
|
315
|
+
* E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
|
|
316
|
+
* supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
|
|
317
|
+
* type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
|
|
318
|
+
* (no prior evidence), so the first observation never moves wealth.
|
|
319
|
+
*
|
|
320
|
+
* `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
|
|
321
|
+
* wealth keeps updating after the crossing (the e-process remains valid), but
|
|
322
|
+
* the decision time is the first crossing.
|
|
323
|
+
*/
|
|
324
|
+
declare function eProcess(opts?: EProcessOptions): EProcess;
|
|
325
|
+
//#endregion
|
|
326
|
+
//#region src/pareto.d.ts
|
|
327
|
+
/**
|
|
328
|
+
* Pareto frontier — multi-objective optimization over candidate runs.
|
|
329
|
+
*
|
|
330
|
+
* Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
|
|
331
|
+
* trading off (cost, latency, quality) or (passRate, tokenBudget,
|
|
332
|
+
* ttfb), you rarely have a single "winner" — you have a set of
|
|
333
|
+
* non-dominated candidates. This module exposes:
|
|
334
|
+
*
|
|
335
|
+
* - `paretoFrontier`: filter a set of candidates to the non-dominated ones
|
|
336
|
+
* - `dominates`: does A dominate B across all objectives?
|
|
337
|
+
*
|
|
338
|
+
* Each objective is declared with a direction: 'maximize' (higher=better)
|
|
339
|
+
* or 'minimize' (lower=better). Candidates are any object; pass an
|
|
340
|
+
* `objective(candidate)` accessor.
|
|
341
|
+
*/
|
|
342
|
+
type Direction = 'maximize' | 'minimize';
|
|
343
|
+
interface Objective<T> {
|
|
344
|
+
/** Stable label used in reports. */
|
|
345
|
+
name: string;
|
|
346
|
+
direction: Direction;
|
|
347
|
+
value: (candidate: T) => number;
|
|
348
|
+
}
|
|
349
|
+
interface ParetoResult<T> {
|
|
350
|
+
frontier: T[];
|
|
351
|
+
dominated: T[];
|
|
352
|
+
/** Index map: frontier[i] dominates each of dominatedBy[i]. */
|
|
353
|
+
dominanceMap: Array<{
|
|
354
|
+
dominator: T;
|
|
355
|
+
dominated: T[];
|
|
356
|
+
}>;
|
|
357
|
+
}
|
|
358
|
+
/** Does candidate A weakly dominate B — strictly better on at least one objective and no worse on any? */
|
|
359
|
+
declare function dominates<T>(a: T, b: T, objectives: Objective<T>[]): boolean;
|
|
360
|
+
/**
|
|
361
|
+
* Compute the non-dominated frontier. Candidates with NaN/Infinity on any
|
|
362
|
+
* objective are excluded (can't rank them). A candidate enters the frontier
|
|
363
|
+
* iff no other candidate dominates it.
|
|
364
|
+
*/
|
|
365
|
+
declare function paretoFrontier<T>(candidates: T[], objectives: Objective<T>[]): ParetoResult<T>;
|
|
366
|
+
//#endregion
|
|
367
|
+
//#region src/campaign/gates/promotion-policy.d.ts
|
|
368
|
+
/** Where an objective's per-cell scalar comes from. `composite` reads the
|
|
369
|
+
* judge's composite; `dimension` reads a named per-dimension score. */
|
|
370
|
+
type ObjectiveSource = {
|
|
371
|
+
kind: 'composite';
|
|
372
|
+
} | {
|
|
373
|
+
kind: 'dimension';
|
|
374
|
+
dimension: string;
|
|
375
|
+
};
|
|
376
|
+
interface PromotionObjective {
|
|
377
|
+
/** Stable label used in reports + `contributingGates`. */
|
|
378
|
+
name: string;
|
|
379
|
+
source: ObjectiveSource;
|
|
380
|
+
/** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients
|
|
381
|
+
* the paired delta so a positive bootstrap always means "candidate better". */
|
|
382
|
+
direction: Direction;
|
|
383
|
+
/** The good-direction paired-delta CI lower bound must EXCEED this to count
|
|
384
|
+
* as a significant gain on this axis. Interpreted in the judge's native
|
|
385
|
+
* scale. Default 0 (⇒ "confidently better"). */
|
|
386
|
+
gainThreshold?: number;
|
|
387
|
+
/** A floor breach (regression) is declared when the good-direction CI lower
|
|
388
|
+
* bound is below −floorTolerance, or when the exact small-sample test proves
|
|
389
|
+
* a drop past it. When omitted it auto-scales off observed magnitudes
|
|
390
|
+
* (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */
|
|
391
|
+
floorTolerance?: number;
|
|
392
|
+
}
|
|
393
|
+
/** Per-axis verdict from the good-direction paired bootstrap. */
|
|
394
|
+
type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs';
|
|
395
|
+
interface AxisEvidence {
|
|
396
|
+
name: string;
|
|
397
|
+
source: ObjectiveSource;
|
|
398
|
+
direction: Direction;
|
|
399
|
+
/** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):
|
|
400
|
+
* a positive value means the candidate is better on this axis.
|
|
401
|
+
*
|
|
402
|
+
* DIAGNOSTIC on a pass/fail axis: there the verdict is decided on Tango's
|
|
403
|
+
* score interval instead, because a percentile bootstrap over a three-atom
|
|
404
|
+
* delta lattice is not a valid interval at the nonzero margin `floorTolerance`
|
|
405
|
+
* and `gainThreshold` create. `ci` carries the interval that decided. */
|
|
406
|
+
bootstrap: PairedBootstrapResult;
|
|
407
|
+
/** Which paired statistic `bootstrap.low`/`.high` bracket. `'mean'` unless the
|
|
408
|
+
* caller asked for the median — on a pass/fail axis the median and its whole
|
|
409
|
+
* CI are pinned at 0 by tie domination and can see neither a gain nor a
|
|
410
|
+
* regression. `bootstrap.median` still carries the median point estimate. */
|
|
411
|
+
bootstrapStatistic: 'median' | 'mean';
|
|
412
|
+
/** The interval the axis verdict was actually decided on, good-direction and
|
|
413
|
+
* in the axis's native units. */
|
|
414
|
+
ci: {
|
|
415
|
+
low: number;
|
|
416
|
+
high: number;
|
|
417
|
+
};
|
|
418
|
+
/** Which estimator produced `ci`. */
|
|
419
|
+
decisionStatistic: PairedDecisionStatistic;
|
|
420
|
+
/** McNemar's exact evidence on a pass/fail axis; null otherwise. */
|
|
421
|
+
mcnemar: PairedMcNemarEvidence | null;
|
|
422
|
+
/** `ci` has zero width — no evidence in either direction, so the axis is
|
|
423
|
+
* neither improved nor regressed however the point estimate sits. */
|
|
424
|
+
indeterminate: boolean;
|
|
425
|
+
/** Paired observations contributing to this axis. */
|
|
426
|
+
n: number;
|
|
427
|
+
minimumRequired: number;
|
|
428
|
+
decisionMethod: PairedDecisionMethod;
|
|
429
|
+
gainThreshold: number;
|
|
430
|
+
floorTolerance: number;
|
|
431
|
+
verdict: AxisVerdict;
|
|
432
|
+
}
|
|
433
|
+
interface EvidenceVector {
|
|
434
|
+
/** One entry per objective — NOTHING averaged across axes. */
|
|
435
|
+
axes: AxisEvidence[];
|
|
436
|
+
/** Smallest paired n across axes that produced observations — the binding
|
|
437
|
+
* evidence-sufficiency constraint. 0 when no axis produced observations. */
|
|
438
|
+
minN: number;
|
|
439
|
+
/** Aggregate per-side cost from the gate context (a constraint input, not a
|
|
440
|
+
* CI axis — see the module header). */
|
|
441
|
+
cost: {
|
|
442
|
+
candidate: number;
|
|
443
|
+
baseline: number;
|
|
444
|
+
};
|
|
445
|
+
}
|
|
446
|
+
/** A promotion strategy: a pure function from the evidence vector to a verdict.
|
|
447
|
+
* Many policies can run over the same `EvidenceVector` and disagree — that's
|
|
448
|
+
* the point (competing strategies, shared evidence). */
|
|
449
|
+
type PromotionPolicy = (ev: EvidenceVector) => GateResult;
|
|
450
|
+
interface BuildEvidenceVectorOptions {
|
|
451
|
+
/** Minimum paired observations before an axis can claim significance; below
|
|
452
|
+
* it the axis is `few_runs`. The exact small-sample test may require more
|
|
453
|
+
* observations at the selected confidence. */
|
|
454
|
+
minProductiveRuns?: number;
|
|
455
|
+
/** Confidence level for every axis bootstrap. Default 0.95. */
|
|
456
|
+
confidence?: number;
|
|
457
|
+
/** Bootstrap resamples. Default 2000. */
|
|
458
|
+
resamples?: number;
|
|
459
|
+
/** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */
|
|
460
|
+
seed?: number;
|
|
461
|
+
/** Paired statistic every axis CI is computed on. Default `'mean'` — see
|
|
462
|
+
* {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */
|
|
463
|
+
statistic?: 'mean' | 'median';
|
|
464
|
+
}
|
|
465
|
+
/**
|
|
466
|
+
* The Evidence Bus. For each objective, pair candidate vs baseline by full
|
|
467
|
+
* cellId and bootstrap a CI on the good-direction paired delta. Reuses the
|
|
468
|
+
* exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
|
|
469
|
+
* a single source of truth governs pairing granularity + scale handling.
|
|
470
|
+
*/
|
|
471
|
+
declare function buildEvidenceVector<TArtifact, TScenario extends Scenario>(ctx: GateContext<TArtifact, TScenario>, objectives: PromotionObjective[], opts?: BuildEvidenceVectorOptions): EvidenceVector;
|
|
472
|
+
/**
|
|
473
|
+
* The default strategy: symmetric multi-objective Pareto significance. Ship iff
|
|
474
|
+
* the candidate weakly dominates the baseline at the confidence level — no axis
|
|
475
|
+
* credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
|
|
476
|
+
* (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
|
|
477
|
+
* need_more_work. Statistically equivalent → hold (never ship noise).
|
|
478
|
+
*/
|
|
479
|
+
declare const paretoPolicy: PromotionPolicy;
|
|
480
|
+
interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
|
|
481
|
+
/** The objective vector. Every axis is both a gain source and a safety floor. */
|
|
482
|
+
objectives: PromotionObjective[];
|
|
483
|
+
/** Strategy applied to the evidence vector. Default `paretoPolicy`. Override
|
|
484
|
+
* to run a stricter/looser strategy over the SAME bus (competing policies). */
|
|
485
|
+
policy?: PromotionPolicy;
|
|
486
|
+
/** Override the gate name in reports. */
|
|
487
|
+
name?: string;
|
|
488
|
+
}
|
|
489
|
+
/**
|
|
490
|
+
* Wrap the bus + a policy as a `Gate`. Plugs into the existing
|
|
491
|
+
* `runImprovementLoop({ gate })` slot and composes via `composeGate`; default
|
|
492
|
+
* loop behavior is unchanged because consumers opt in by passing this gate.
|
|
493
|
+
*/
|
|
494
|
+
declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: ParetoSignificanceGateOptions): Gate<TArtifact, TScenario>;
|
|
495
|
+
//#endregion
|
|
496
|
+
//#region src/campaign/gates/power-preflight.d.ts
|
|
497
|
+
/**
|
|
498
|
+
* Power preflight — "can this budget detect the effect you are hunting?"
|
|
499
|
+
*
|
|
500
|
+
* The failure it prevents (measured, twice): a live prompt-improvement campaign ran
|
|
501
|
+
* 333 sandbox cells over 5.6 hours and produced a +0.08 holdout lift the ship gate
|
|
502
|
+
* (paired bootstrap, CI.low > 0.05) could not distinguish from zero — because at
|
|
503
|
+
* that holdout size and worker variance the MINIMUM DETECTABLE lift was larger than
|
|
504
|
+
* any effect a prompt change plausibly produces. The budget was spent learning what
|
|
505
|
+
* a 30-second calculation on the baseline cells already knew. No eval framework we
|
|
506
|
+
* know of surfaces this; every underpowered improvement run everywhere ends in an
|
|
507
|
+
* uninformative "hold".
|
|
508
|
+
*
|
|
509
|
+
* Model: the ship rule is `CI.low(paired Δ) > deltaThreshold`. Approximating the
|
|
510
|
+
* bootstrap CI as normal, `CI.low ≈ effect − z·sd_Δ/√n`, so the smallest shippable
|
|
511
|
+
* true effect is `MDE = deltaThreshold + z·sd_Δ/√n`. The paired-delta SD is unknown
|
|
512
|
+
* before the candidate exists; we bound it by the zero-correlation case
|
|
513
|
+
* `sd_Δ ≤ √2·sd_baseline` — a CONSERVATIVE (upper) MDE, which is the correct
|
|
514
|
+
* direction for a warning. Pairing is per cell (`scenario:rep`), so reps multiply n.
|
|
515
|
+
*
|
|
516
|
+
* Standalone by design: feed it any baseline composites (a `gate:'none'` run, a
|
|
517
|
+
* live-proof table) BEFORE budgeting the real search; `selfImprove` also attaches
|
|
518
|
+
* it to every result and warns when the run was structurally unable to ship.
|
|
519
|
+
*/
|
|
520
|
+
interface PowerPreflightOptions {
|
|
521
|
+
/** Per-cell baseline composites on the HOLDOUT scenarios (one per scenario:rep cell). */
|
|
522
|
+
baselineComposites: number[];
|
|
523
|
+
/** Paired observations the budgeted comparison will produce
|
|
524
|
+
* (holdout scenarios × reps). Defaults to `baselineComposites.length`. */
|
|
525
|
+
pairedN?: number;
|
|
526
|
+
/** The ship gate's effect-size threshold. Default 0.05 (defaultProductionGate). */
|
|
527
|
+
deltaThreshold?: number;
|
|
528
|
+
/** CI confidence the gate uses. Default 0.95. */
|
|
529
|
+
confidence?: number;
|
|
530
|
+
/** True when the holdout is scored by the SAME judge/scorer family as the gate
|
|
531
|
+
* (selfImprove's default composition — one judge scores everything). Under a
|
|
532
|
+
* shared channel, raising paired n reduces only the IDIOSYNCRATIC noise share;
|
|
533
|
+
* systematic judge bias is untouched, so the MDE here is a lower bound and the
|
|
534
|
+
* only full debiaser is an independent second scoring channel
|
|
535
|
+
* (recursive-self-improvement S1c, closed form in EXP-023 P0). Default false. */
|
|
536
|
+
sharedScorerChannel?: boolean;
|
|
537
|
+
}
|
|
538
|
+
interface PowerPreflight {
|
|
539
|
+
/** Paired observations the comparison will have. */
|
|
540
|
+
n: number;
|
|
541
|
+
/** Baseline per-cell composite standard deviation (the variance the effect must beat). */
|
|
542
|
+
sd: number;
|
|
543
|
+
/** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */
|
|
544
|
+
mde: number;
|
|
545
|
+
/** Baseline holdout composite mean. */
|
|
546
|
+
baselineMean: number;
|
|
547
|
+
/** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */
|
|
548
|
+
headroom: number;
|
|
549
|
+
/** True when even the largest achievable effect (headroom) is below the MDE —
|
|
550
|
+
* the run is structurally unable to ship regardless of proposal quality.
|
|
551
|
+
* Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */
|
|
552
|
+
underpowered: boolean;
|
|
553
|
+
/** True when composites look [0,1]-scaled; headroom/underpowered are only
|
|
554
|
+
* meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */
|
|
555
|
+
scaleAssumed: boolean;
|
|
556
|
+
deltaThreshold: number;
|
|
557
|
+
confidence: number;
|
|
558
|
+
/** Set when the holdout shares the gate's scoring channel: more cells cannot
|
|
559
|
+
* buy back systematic judge bias — treat the MDE as a lower bound. */
|
|
560
|
+
sharedChannelCaveat?: string;
|
|
561
|
+
/** One actionable sentence for humans and logs. */
|
|
562
|
+
recommendation: string;
|
|
563
|
+
}
|
|
564
|
+
/** Estimate the minimum detectable lift a paired-holdout improvement run can
|
|
565
|
+
* ship at a given budget, from the baseline holdout composites — call it BEFORE
|
|
566
|
+
* spending a search to learn whether the effect you are hunting is even
|
|
567
|
+
* observable at this holdout size and worker variance. */
|
|
568
|
+
declare function powerPreflight(opts: PowerPreflightOptions): PowerPreflight;
|
|
569
|
+
//#endregion
|
|
570
|
+
//#region src/paired-arms.d.ts
|
|
571
|
+
/** One arm observation of one work item. Structural on purpose: callers
|
|
572
|
+
* project their own record type (e.g. a `RunRecord`) into this shape. */
|
|
573
|
+
interface PairedArmRow {
|
|
574
|
+
/** Matching key — rows sharing a `pairKey` across both arms form pairs
|
|
575
|
+
* (typically the task/scenario/seed identity). */
|
|
576
|
+
pairKey: string;
|
|
577
|
+
/** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on
|
|
578
|
+
* every row of a `pairKey` that has more than one rep in either arm; reps
|
|
579
|
+
* then pair only on exact (`pairKey`, `repKey`) match, never on outcome
|
|
580
|
+
* content. Optional when each arm has at most one rep of the item. */
|
|
581
|
+
repKey?: string;
|
|
582
|
+
/** Arm label this row was produced under. */
|
|
583
|
+
arm: string;
|
|
584
|
+
/** Binary outcome; omit when the comparison has no pass/fail notion. */
|
|
585
|
+
pass?: boolean;
|
|
586
|
+
/** Named numeric measurements (score, cost, latency, …). */
|
|
587
|
+
metrics?: Record<string, number>;
|
|
588
|
+
}
|
|
589
|
+
interface PairArmsOptions {
|
|
590
|
+
/** Arm treated as the control side of every pair. */
|
|
591
|
+
baselineArm: string;
|
|
592
|
+
/** Arm treated as the treatment side of every pair. */
|
|
593
|
+
treatmentArm: string;
|
|
594
|
+
}
|
|
595
|
+
/** One matched (baseline, treatment) observation of the same work item. */
|
|
596
|
+
interface MatchedPair {
|
|
597
|
+
pairKey: string;
|
|
598
|
+
/** 0-based position of this pair within its `pairKey`, ordered by sorted
|
|
599
|
+
* `repKey` (always 0 for a single-rep item). The rep identity itself is on
|
|
600
|
+
* the rows (`baseline.repKey` / `treatment.repKey`). */
|
|
601
|
+
repIndex: number;
|
|
602
|
+
baseline: PairedArmRow;
|
|
603
|
+
treatment: PairedArmRow;
|
|
604
|
+
}
|
|
605
|
+
interface PairArmsResult {
|
|
606
|
+
/** Matched pairs, ordered by (`pairKey`, `repIndex`). */
|
|
607
|
+
pairs: MatchedPair[];
|
|
608
|
+
/** Baseline rows left without a treatment counterpart — reported, never
|
|
609
|
+
* silently dropped. */
|
|
610
|
+
unpairedBaseline: PairedArmRow[];
|
|
611
|
+
/** Treatment rows left without a baseline counterpart. */
|
|
612
|
+
unpairedTreatment: PairedArmRow[];
|
|
613
|
+
}
|
|
614
|
+
/**
|
|
615
|
+
* Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
|
|
616
|
+
*
|
|
617
|
+
* A `pairKey` with at most one row per arm pairs directly, no `repKey`
|
|
618
|
+
* needed. A `pairKey` with multiple reps in either arm requires `repKey` on
|
|
619
|
+
* every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
|
|
620
|
+
* match — pairing is keyed purely on row identity, never on outcome content
|
|
621
|
+
* (outcome-keyed matching deflates discordant counts and biases McNemar), and
|
|
622
|
+
* is therefore independent of input order. Reps whose `repKey` has no
|
|
623
|
+
* counterpart in the other arm, and items present in only one arm, land in
|
|
624
|
+
* the unpaired lists — reported, never truncated.
|
|
625
|
+
*
|
|
626
|
+
* Fail-loud: throws when either named arm has zero rows (an unknown arm
|
|
627
|
+
* name would otherwise read as "everything unpaired"), when the two arm
|
|
628
|
+
* names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
|
|
629
|
+
* when a (`pairKey`, arm) group repeats a `repKey` (the match would be
|
|
630
|
+
* ambiguous).
|
|
631
|
+
*/
|
|
632
|
+
declare function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult;
|
|
633
|
+
/** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */
|
|
634
|
+
interface PairedCorrectness {
|
|
635
|
+
/** Discordant pairs where the treatment passed and the baseline failed. */
|
|
636
|
+
b10: number;
|
|
637
|
+
/** Discordant pairs where the baseline passed and the treatment failed. */
|
|
638
|
+
b01: number;
|
|
639
|
+
/** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */
|
|
640
|
+
mcnemar: McNemarResult;
|
|
641
|
+
/** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */
|
|
642
|
+
riskDifference: RiskDifferenceResult;
|
|
643
|
+
}
|
|
644
|
+
/** Paired delta summary for one named metric (delta = treatment − baseline). */
|
|
645
|
+
interface PairedMetricDelta {
|
|
646
|
+
name: string;
|
|
647
|
+
/** Pairs where BOTH sides carry a finite value for this metric. */
|
|
648
|
+
n: number;
|
|
649
|
+
/** Pairs where at least one side does not carry the metric. */
|
|
650
|
+
nMissing: number;
|
|
651
|
+
/** Median paired delta, or null when `n === 0`. */
|
|
652
|
+
medianDelta: number | null;
|
|
653
|
+
/** Mean paired delta, or null when `n === 0`. */
|
|
654
|
+
meanDelta: number | null;
|
|
655
|
+
/** Bootstrap CI on the paired delta (`pairedBootstrap`); null when
|
|
656
|
+
* `n === 0` — a zero-width [0, 0] interval on no data would read as a
|
|
657
|
+
* measured tight null. */
|
|
658
|
+
bootstrapCi: PairedBootstrapResult | null;
|
|
659
|
+
/** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */
|
|
660
|
+
wilcoxon: {
|
|
661
|
+
w: number;
|
|
662
|
+
p: number;
|
|
663
|
+
} | null;
|
|
664
|
+
}
|
|
665
|
+
interface ComparePairedArmsOptions extends PairArmsOptions {
|
|
666
|
+
/** Metrics to compare. Default: every metric name observed on any matched
|
|
667
|
+
* pair, sorted. A name that appears on no pair is still reported (with
|
|
668
|
+
* `n = 0`) so a misspelled metric is visible instead of vanishing. */
|
|
669
|
+
metricNames?: string[];
|
|
670
|
+
/** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */
|
|
671
|
+
bootstrap?: PairedBootstrapOptions;
|
|
672
|
+
}
|
|
673
|
+
interface PairedArmsComparison {
|
|
674
|
+
nPairs: number;
|
|
675
|
+
nUnpairedBaseline: number;
|
|
676
|
+
nUnpairedTreatment: number;
|
|
677
|
+
/** null when no matched pair carries `pass` on both sides — a pass/fail
|
|
678
|
+
* verdict over rows that never measured pass/fail would be fabricated. */
|
|
679
|
+
correctness: PairedCorrectness | null;
|
|
680
|
+
metricDeltas: PairedMetricDelta[];
|
|
681
|
+
}
|
|
682
|
+
/**
|
|
683
|
+
* Full matched-pair arm comparison: pair via {@link pairArms}, then compose
|
|
684
|
+
* the paired estimators from `statistics` over the matched pairs.
|
|
685
|
+
*
|
|
686
|
+
* Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
|
|
687
|
+
* is that subset's size); each metric uses only the pairs where both sides
|
|
688
|
+
* carry a finite value for it, with the remainder counted in `nMissing`.
|
|
689
|
+
* Deltas are treatment − baseline throughout.
|
|
690
|
+
*
|
|
691
|
+
* Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
|
|
692
|
+
* non-finite metric value — silently treating corrupt telemetry as "metric
|
|
693
|
+
* absent" would misreport it as missing coverage.
|
|
694
|
+
*/
|
|
695
|
+
declare function comparePairedArms(rows: readonly PairedArmRow[], opts: ComparePairedArmsOptions): PairedArmsComparison;
|
|
696
|
+
interface MatchedRunRecordPair {
|
|
697
|
+
pairKey: string;
|
|
698
|
+
repKey: string;
|
|
699
|
+
baseline: RunRecord;
|
|
700
|
+
treatment: RunRecord;
|
|
701
|
+
}
|
|
702
|
+
interface PairRunRecordsResult {
|
|
703
|
+
pairs: MatchedRunRecordPair[];
|
|
704
|
+
unpairedBaseline: RunRecord[];
|
|
705
|
+
unpairedTreatment: RunRecord[];
|
|
706
|
+
}
|
|
707
|
+
/**
|
|
708
|
+
* Pair two RunRecord arms by the identity of the evaluated work:
|
|
709
|
+
* `(experimentId, scenarioId, seed)`.
|
|
710
|
+
*
|
|
711
|
+
* Falling back to array order, candidate id, or experiment id can compare
|
|
712
|
+
* different tasks and fabricate lift. Duplicate identities throw.
|
|
713
|
+
*/
|
|
714
|
+
declare function pairRunRecords(baselineRuns: readonly RunRecord[], treatmentRuns: readonly RunRecord[]): PairRunRecordsResult;
|
|
715
|
+
//#endregion
|
|
716
|
+
//#region src/pre-registration.d.ts
|
|
717
|
+
/**
|
|
718
|
+
* Pre-registered hypotheses — declare what you're testing BEFORE the
|
|
719
|
+
* run, check it AFTER. Prevents p-hacking, optional stopping, and the
|
|
720
|
+
* "we ran until it looked good" failure mode.
|
|
721
|
+
*
|
|
722
|
+
* Manifest is a plain JSON-friendly object. Sign it with a content hash
|
|
723
|
+
* + timestamp; the registered record becomes immutable. Post-run,
|
|
724
|
+
* evaluate the manifest against observed results — the library refuses
|
|
725
|
+
* to let you re-interpret a different metric as the declared one.
|
|
726
|
+
*
|
|
727
|
+
* A signed manifest is a portable record: it is written once and verified
|
|
728
|
+
* later, possibly by a different release. `algo` names the digest scheme it
|
|
729
|
+
* was signed under, and verification selects the encoder by that field, so a
|
|
730
|
+
* manifest signed by an earlier release still verifies.
|
|
731
|
+
*/
|
|
732
|
+
interface HypothesisManifest {
|
|
733
|
+
id: string;
|
|
734
|
+
/** Human prose — goes into the audit trail. */
|
|
735
|
+
hypothesis: string;
|
|
736
|
+
/** Metric the hypothesis claims to move. */
|
|
737
|
+
metric: string;
|
|
738
|
+
/** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
|
|
739
|
+
direction: 'increase' | 'decrease';
|
|
740
|
+
/** Minimum effect size to count (same units as the metric). */
|
|
741
|
+
minEffect: number;
|
|
742
|
+
/** Alpha threshold. */
|
|
743
|
+
alpha: number;
|
|
744
|
+
/** Target statistical power at which sample size was pre-computed. */
|
|
745
|
+
power: number;
|
|
746
|
+
/** Declared N per arm before running. */
|
|
747
|
+
preRegisteredN: number;
|
|
748
|
+
/** ISO8601 timestamp the manifest was registered. */
|
|
749
|
+
registeredAt: string;
|
|
750
|
+
/** Optional identifiers to tie into the trace corpus. */
|
|
751
|
+
baselineLabel?: string;
|
|
752
|
+
candidateLabel?: string;
|
|
753
|
+
}
|
|
754
|
+
/**
|
|
755
|
+
* Identifier for the hashing scheme used to produce `contentHash`.
|
|
756
|
+
*
|
|
757
|
+
* Both schemes are sha256 hex over the manifest with `contentHash` and `algo`
|
|
758
|
+
* stripped, and differ only in how that manifest is serialized:
|
|
759
|
+
*
|
|
760
|
+
* - `'sha256-rfc8785'` — RFC 8785 canonical JSON. What {@link signManifest}
|
|
761
|
+
* emits.
|
|
762
|
+
* - `'sha256-content'` — key-sorted `JSON.stringify`. Read-only: manifests
|
|
763
|
+
* signed by an earlier release carry it, or carry no `algo` at all, and
|
|
764
|
+
* {@link verifyManifest} still verifies them.
|
|
765
|
+
*/
|
|
766
|
+
type SignedManifestAlgo = 'sha256-content' | 'sha256-rfc8785';
|
|
767
|
+
interface SignedManifest extends HypothesisManifest {
|
|
768
|
+
/** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
|
|
769
|
+
contentHash: string;
|
|
770
|
+
/**
|
|
771
|
+
* Algorithm string describing how `contentHash` was produced.
|
|
772
|
+
*
|
|
773
|
+
* Optional on the type so serialized manifests without it still parse,
|
|
774
|
+
* but ALWAYS populated by {@link signManifest}. Consumers that want to
|
|
775
|
+
* enforce a known algorithm should reject manifests where this field
|
|
776
|
+
* is missing or unrecognized.
|
|
777
|
+
*/
|
|
778
|
+
algo?: SignedManifestAlgo;
|
|
779
|
+
}
|
|
780
|
+
interface HypothesisResult {
|
|
781
|
+
manifest: SignedManifest;
|
|
782
|
+
observedN: number;
|
|
783
|
+
observedEffect: number;
|
|
784
|
+
observedPValue: number;
|
|
785
|
+
/** True iff the observed effect hits the pre-declared direction with
|
|
786
|
+
* magnitude ≥ minEffect AND p < alpha. */
|
|
787
|
+
confirmed: boolean;
|
|
788
|
+
/** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
|
|
789
|
+
rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
|
|
790
|
+
notes?: string;
|
|
791
|
+
}
|
|
792
|
+
/**
|
|
793
|
+
* SHA-256 hex (full 64 chars) over the RFC 8785 canonical JSON encoding of
|
|
794
|
+
* `obj` — the package's one identity scheme, shared with `ledger-core`.
|
|
795
|
+
*
|
|
796
|
+
* Values canonical JSON cannot represent faithfully — `undefined`, `NaN`,
|
|
797
|
+
* class instances, cycles — are refused rather than coerced, because a
|
|
798
|
+
* coercion maps two distinct records onto one digest.
|
|
799
|
+
*
|
|
800
|
+
* Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
|
|
801
|
+
* which takes a string input and returns a truncated 12-char prompt id.
|
|
802
|
+
*
|
|
803
|
+
* @example
|
|
804
|
+
* const hash = await hashJson({ id: '1', kind: 'spec' })
|
|
805
|
+
* // 'a3f1...' (64 hex chars)
|
|
806
|
+
*/
|
|
807
|
+
declare function hashJson<T>(obj: T): Promise<string>;
|
|
808
|
+
/**
|
|
809
|
+
* Digest of a manifest under its own declared scheme, with `contentHash` and
|
|
810
|
+
* `algo` stripped. Synchronous, so a caller that must fail before consuming an
|
|
811
|
+
* observation does not have to await. Throws on an `algo` this release does
|
|
812
|
+
* not know — an unverifiable manifest must not read as a valid one.
|
|
813
|
+
*/
|
|
814
|
+
declare function manifestContentDigest(manifest: SignedManifest): string;
|
|
815
|
+
/**
|
|
816
|
+
* Sign a manifest with a SHA-256 content hash over its RFC 8785 canonical
|
|
817
|
+
* JSON, with `contentHash` and `algo` stripped, and stamp the scheme in
|
|
818
|
+
* `algo` so a later reader knows which encoder to verify with.
|
|
819
|
+
*/
|
|
820
|
+
declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
|
|
821
|
+
/**
|
|
822
|
+
* Verify that a signed manifest has not been tampered with, under the scheme
|
|
823
|
+
* the manifest itself declares.
|
|
824
|
+
*/
|
|
825
|
+
declare function verifyManifest(m: SignedManifest): Promise<boolean>;
|
|
826
|
+
/**
|
|
827
|
+
* Evaluate a pre-registered hypothesis against observed results.
|
|
828
|
+
* Mechanical — no re-interpretation permitted.
|
|
829
|
+
*/
|
|
830
|
+
declare function evaluateHypothesis(manifest: SignedManifest, observed: {
|
|
831
|
+
n: number;
|
|
832
|
+
effect: number;
|
|
833
|
+
pValue: number;
|
|
834
|
+
}): Promise<HypothesisResult>;
|
|
835
|
+
//#endregion
|
|
836
|
+
//#region src/campaign/gates/sequential.d.ts
|
|
837
|
+
type SequentialDecision = 'promote' | 'continue' | 'undecided-at-maxN';
|
|
838
|
+
interface SequentialObservation {
|
|
839
|
+
decision: SequentialDecision;
|
|
840
|
+
/** Current e-value (the betting wealth) against H0. */
|
|
841
|
+
eValue: number;
|
|
842
|
+
/** Paired deltas consumed so far. */
|
|
843
|
+
n: number;
|
|
844
|
+
/** Names the decision basis. For 'undecided-at-maxN' it states explicitly
|
|
845
|
+
* that exhausting the budget is NOT evidence of no effect. */
|
|
846
|
+
reason: string;
|
|
847
|
+
}
|
|
848
|
+
interface SequentialPairedGateOptions {
|
|
849
|
+
/** Type-I budget. With `preRegistration` bound this MUST match
|
|
850
|
+
* `manifest.alpha` (conflict throws). Default 0.05. */
|
|
851
|
+
alpha?: number;
|
|
852
|
+
/** Minimum paired deltas before a promote may fire. The stopping rule is
|
|
853
|
+
* "first n ≥ minN with e-value ≥ 1/alpha" — still a valid stopping time.
|
|
854
|
+
* Default 5. */
|
|
855
|
+
minN?: number;
|
|
856
|
+
/** Pre-registered observation budget. Required unless `preRegistration`
|
|
857
|
+
* supplies it via `preRegisteredN` (conflict throws). */
|
|
858
|
+
maxN?: number;
|
|
859
|
+
/** Bet truncation forwarded to `eProcess`. Default 0.5. */
|
|
860
|
+
maxBet?: number;
|
|
861
|
+
/** Bound on |delta| in the judge's native scale; deltas are mapped to
|
|
862
|
+
* x = (d/scale + 1)/2 ∈ [0,1]. A delta outside ±scale throws (use
|
|
863
|
+
* `detectScale` to pick 1 vs 100 BEFORE streaming). Default 1. */
|
|
864
|
+
scale?: number;
|
|
865
|
+
/** Seed for the data-independent shuffle of paired deltas in `decide(ctx)`
|
|
866
|
+
* (exchangeability guard). Default 1337. */
|
|
867
|
+
shuffleSeed?: number;
|
|
868
|
+
/** Bind the pre-registered hypothesis. Verified (content hash) at
|
|
869
|
+
* construction; alpha/maxN/direction/minEffect come FROM the manifest. */
|
|
870
|
+
preRegistration?: SignedManifest;
|
|
871
|
+
/** Override the gate name in reports. */
|
|
872
|
+
name?: string;
|
|
873
|
+
/** Continue the observe-stream from a `state()` snapshot taken before a
|
|
874
|
+
* restart. The gate's parameters still resolve from this options object
|
|
875
|
+
* (and the manifest); the snapshot must have been recorded under the same
|
|
876
|
+
* alpha, maxBet, and null boundary, and its decision must be one this
|
|
877
|
+
* configuration could have reached at its n, or construction throws.
|
|
878
|
+
* `decide(ctx)` is unaffected — it always runs on a fresh stream. */
|
|
879
|
+
resume?: SequentialStreamState;
|
|
880
|
+
}
|
|
881
|
+
/** Snapshot of one observe-stream: the e-process state plus the gate's own
|
|
882
|
+
* decision, which applies `minN` and `maxN` on top of the core latch. A
|
|
883
|
+
* gate rebuilt with `resume` from this value continues exactly where the
|
|
884
|
+
* snapshot was taken. */
|
|
885
|
+
type SequentialStreamState = EProcessState & {
|
|
886
|
+
decision: SequentialDecision;
|
|
887
|
+
};
|
|
888
|
+
interface SequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario> extends Gate<TArtifact, TScenario> {
|
|
889
|
+
/** Streaming entry point: feed one paired per-scenario delta
|
|
890
|
+
* (candidate − baseline, native scale). Each gate instance carries ONE
|
|
891
|
+
* observe-stream; `decide(ctx)` runs on its own fresh stream and never
|
|
892
|
+
* consumes or advances this one. 'promote' is sticky; observing past the
|
|
893
|
+
* pre-registered maxN throws (extending a finished stream after seeing
|
|
894
|
+
* the result reopens optional stopping — start a NEW pre-registered
|
|
895
|
+
* test). */
|
|
896
|
+
observe(delta: number): SequentialObservation;
|
|
897
|
+
/** Read-only snapshot of the observe-stream. Pass it back as
|
|
898
|
+
* `resume` to continue the stream after a restart. */
|
|
899
|
+
state(): SequentialStreamState;
|
|
900
|
+
}
|
|
901
|
+
/**
|
|
902
|
+
* Anytime-valid sequential paired gate. Conforms to the existing `Gate`
|
|
903
|
+
* contract (`decide(ctx)` consumes candidate vs baseline judge scores via
|
|
904
|
+
* `pairHoldout` — same pairing granularity as the fixed-n gates: full cellId,
|
|
905
|
+
* never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
|
|
906
|
+
* that score cells incrementally and want to stop mid-stream.
|
|
907
|
+
*
|
|
908
|
+
* Decision mapping onto the substrate's five-valued `GateDecision`:
|
|
909
|
+
* - 'promote' → 'ship'
|
|
910
|
+
* - 'continue' → 'need_more_work' (stream ended before maxN with
|
|
911
|
+
* the e-value undecided — more reps could decide)
|
|
912
|
+
* - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
|
|
913
|
+
* evidence of no effect (never a silent default)
|
|
914
|
+
*/
|
|
915
|
+
declare function sequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: SequentialPairedGateOptions): SequentialPairedGate<TArtifact, TScenario>;
|
|
916
|
+
interface SequentialDecideOptions {
|
|
917
|
+
/** Type-I budget for the early-stop evidence. Default 0.05. */
|
|
918
|
+
alpha?: number;
|
|
919
|
+
/** Minimum paired deltas before a stop may fire. Default 5. */
|
|
920
|
+
minN?: number;
|
|
921
|
+
/** Bet truncation forwarded to `eProcess`. Default 0.5. */
|
|
922
|
+
maxBet?: number;
|
|
923
|
+
/** Bound on |per-scenario composite delta|. Default 1. */
|
|
924
|
+
scale?: number;
|
|
925
|
+
}
|
|
926
|
+
interface SequentialDecideFn {
|
|
927
|
+
(args: {
|
|
928
|
+
history: GenerationRecord[];
|
|
929
|
+
}): {
|
|
930
|
+
stop: boolean;
|
|
931
|
+
reason?: string;
|
|
932
|
+
};
|
|
933
|
+
/** Read-only snapshot of the accumulated e-process (observability + tests). */
|
|
934
|
+
state(): EProcessState;
|
|
935
|
+
}
|
|
936
|
+
/**
|
|
937
|
+
* `SurfaceProposer.decide` adapter — stops the optimization loop the moment
|
|
938
|
+
* the e-process decides the loop has produced a real improvement, instead of
|
|
939
|
+
* always running `maxGenerations`.
|
|
940
|
+
*
|
|
941
|
+
* Stream: for each generation g ≥ 1, the per-scenario composite deltas of
|
|
942
|
+
* generation g's top candidate vs the generation-0 top candidate (the
|
|
943
|
+
* incumbent the loop set out to beat), paired by scenarioId. H0: no proposed
|
|
944
|
+
* surface improves any scenario's expected composite over the incumbent —
|
|
945
|
+
* under it every delta has conditional mean ≤ 0 and the e-process is valid.
|
|
946
|
+
* Once wealth ≥ 1/alpha the loop stops and hands the winner to the promotion
|
|
947
|
+
* gate (which re-scores on HELD-OUT data — this adapter only spends the
|
|
948
|
+
* exploration budget, it never promotes).
|
|
949
|
+
*
|
|
950
|
+
* Honesty caveats: (1) the incumbent's scores are measured once and shared
|
|
951
|
+
* across all generations' deltas, so type-I control is exact only insofar as
|
|
952
|
+
* those scores approximate the incumbent's true per-scenario means (more reps
|
|
953
|
+
* → tighter); (2) an UNDECIDED process never stops the loop — absence of a
|
|
954
|
+
* crossing is NOT evidence of no effect, so the loop simply runs its normal
|
|
955
|
+
* course. Calling the adapter repeatedly with a growing history consumes each
|
|
956
|
+
* generation exactly once (re-feeding an already-seen record would double-count
|
|
957
|
+
* evidence).
|
|
958
|
+
*/
|
|
959
|
+
declare function sequentialDecide(options?: SequentialDecideOptions): SequentialDecideFn;
|
|
960
|
+
//#endregion
|
|
961
|
+
//#region src/campaign/gates/statistical-heldout.d.ts
|
|
962
|
+
interface PairedHoldout {
|
|
963
|
+
/** Baseline scalar per paired cell (same order as `after`/`cellIds`). */
|
|
964
|
+
before: number[];
|
|
965
|
+
/** Candidate scalar per paired cell. */
|
|
966
|
+
after: number[];
|
|
967
|
+
/** The full cellIds (`scenario:rep`) that paired, in order. */
|
|
968
|
+
cellIds: string[];
|
|
969
|
+
}
|
|
970
|
+
/**
|
|
971
|
+
* Pair candidate vs baseline holdout observations by FULL cellId. `select`
|
|
972
|
+
* pulls the scalar from a cell's judge reports (composite, or a named
|
|
973
|
+
* dimension); a cell contributes the mean of `select` across its judges. Cells
|
|
974
|
+
* whose scenario is not in `scenarioIds`, or where `select` is undefined for
|
|
975
|
+
* every judge on either side, are skipped on BOTH sides so the arrays stay
|
|
976
|
+
* paired. Throws when the two maps disagree on which holdout cells exist — a
|
|
977
|
+
* load-bearing invariant: the baseline + winner holdout campaigns run the same
|
|
978
|
+
* scenarios with the same seed base, so their cellIds MUST align; a mismatch
|
|
979
|
+
* means a silent pairing bug, not a soft fallback.
|
|
980
|
+
*/
|
|
981
|
+
declare function pairHoldout(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, select: (s: JudgeScore) => number | undefined): PairedHoldout;
|
|
982
|
+
interface HeldoutSignificance {
|
|
983
|
+
paired: PairedHoldout;
|
|
984
|
+
/**
|
|
985
|
+
* The paired bootstrap on the requested statistic (MEAN by default — see the
|
|
986
|
+
* tie note on `heldoutSignificance`).
|
|
987
|
+
*
|
|
988
|
+
* DIAGNOSTIC, not necessarily the interval the verdict keyed on. On a
|
|
989
|
+
* two-point (pass/fail) outcome the decision routes to Tango's score interval
|
|
990
|
+
* instead, because a percentile bootstrap of the mean over a three-atom
|
|
991
|
+
* lattice is not a valid interval at a nonzero margin. Read
|
|
992
|
+
* `decision.low`/`decision.high` for the interval that actually decided, and
|
|
993
|
+
* `decisionStatistic` for which one it is.
|
|
994
|
+
*/
|
|
995
|
+
bootstrap: PairedBootstrapResult;
|
|
996
|
+
/** The MEDIAN paired-delta bootstrap, reported as a diagnostic. When many
|
|
997
|
+
* scenarios are tied (both sides solve them), the median is pinned near 0
|
|
998
|
+
* regardless of the mean lift — comparing the two exposes tie-domination. */
|
|
999
|
+
medianBootstrap: PairedBootstrapResult;
|
|
1000
|
+
/**
|
|
1001
|
+
* The full promotion decision: which estimator the outcome's shape admits,
|
|
1002
|
+
* the interval it produced, McNemar's exact veto on the two-point path, and
|
|
1003
|
+
* whether the interval was zero-width (no evidence in either direction). The
|
|
1004
|
+
* single source of `significant`.
|
|
1005
|
+
*/
|
|
1006
|
+
decision: PairedPromotionDecision;
|
|
1007
|
+
/** Which paired estimator the verdict was decided on. */
|
|
1008
|
+
decisionStatistic: PairedDecisionStatistic;
|
|
1009
|
+
/** McNemar's exact evidence on the two-point path; null otherwise. */
|
|
1010
|
+
mcnemar: PairedMcNemarEvidence | null;
|
|
1011
|
+
/** Fraction of paired observations that are exact ties (|delta| < 1e-9). A
|
|
1012
|
+
* high tie fraction is WHY a median-based gate would have missed a real lift;
|
|
1013
|
+
* it is the observability the tie fix adds. */
|
|
1014
|
+
tieFraction: number;
|
|
1015
|
+
/** n paired observations. */
|
|
1016
|
+
n: number;
|
|
1017
|
+
/** Effective minimum after applying the bootstrap's hard statistical floor. */
|
|
1018
|
+
minimumRequired: number;
|
|
1019
|
+
/** Statistical method that carried the decision. */
|
|
1020
|
+
decisionMethod: PairedDecisionMethod;
|
|
1021
|
+
/** Exact one-sided p-value on the small-sample path; otherwise null. */
|
|
1022
|
+
pValue: number | null;
|
|
1023
|
+
/** True iff n >= minimumRequired, the DECIDING interval has nonzero width,
|
|
1024
|
+
* its lower bound clears the threshold, and McNemar's exact test does not
|
|
1025
|
+
* veto at a non-negative threshold. */
|
|
1026
|
+
significant: boolean;
|
|
1027
|
+
/** Set when n < minimumRequired — too little evidence to claim significance. */
|
|
1028
|
+
fewRuns: boolean;
|
|
1029
|
+
}
|
|
1030
|
+
interface HeldoutSignificanceOptions {
|
|
1031
|
+
deltaThreshold?: number;
|
|
1032
|
+
minProductiveRuns?: number;
|
|
1033
|
+
confidence?: number;
|
|
1034
|
+
resamples?: number;
|
|
1035
|
+
/** Fixed by default for a deterministic, reproducible gate verdict. */
|
|
1036
|
+
seed?: number;
|
|
1037
|
+
statistic?: 'mean' | 'median';
|
|
1038
|
+
}
|
|
1039
|
+
/**
|
|
1040
|
+
* Significance of the held-out composite lift: ship only when the lower bound
|
|
1041
|
+
* of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
|
|
1042
|
+
* 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
|
|
1043
|
+
* scale.
|
|
1044
|
+
*
|
|
1045
|
+
* The decision is delegated whole to {@link decidePairedPromotion}, the one
|
|
1046
|
+
* copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
|
|
1047
|
+
* also calls. That module's header carries the measurements; the short version
|
|
1048
|
+
* is three guards a bare `bootstrap.low > threshold` does not have:
|
|
1049
|
+
*
|
|
1050
|
+
* - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
|
|
1051
|
+
* only paired-binary construction that stays valid at a nonzero margin;
|
|
1052
|
+
* - McNemar's exact test VETOES at any non-negative threshold;
|
|
1053
|
+
* - a ZERO-WIDTH interval is refused rather than promoted, in either
|
|
1054
|
+
* direction — [0,0] clears every negative threshold and [g,g] clears every
|
|
1055
|
+
* threshold below g, and both are an absence of evidence, not a result.
|
|
1056
|
+
*
|
|
1057
|
+
* Measured on this function before those guards landed, at a nominal 5 %:
|
|
1058
|
+
* 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
|
|
1059
|
+
* and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
|
|
1060
|
+
* delta is exactly 0.
|
|
1061
|
+
*
|
|
1062
|
+
* At small n, where the percentile bootstrap is descriptive only, a
|
|
1063
|
+
* pre-registered exact sign test still carries the bootstrap path.
|
|
1064
|
+
*/
|
|
1065
|
+
declare function heldoutSignificance(paired: PairedHoldout, opts?: HeldoutSignificanceOptions): HeldoutSignificance;
|
|
1066
|
+
interface DimensionRegression {
|
|
1067
|
+
dimension: string;
|
|
1068
|
+
/** Paired bootstrap on (candidate − baseline). DIAGNOSTIC on a pass/fail
|
|
1069
|
+
* dimension, where `ci` carries the interval that decided instead. */
|
|
1070
|
+
bootstrap: PairedBootstrapResult;
|
|
1071
|
+
/** Which paired statistic `bootstrap.low` is the lower bound of. `'mean'`
|
|
1072
|
+
* unless the caller asked for the median. `bootstrap.median` still carries
|
|
1073
|
+
* the median point estimate either way. */
|
|
1074
|
+
bootstrapStatistic: 'median' | 'mean';
|
|
1075
|
+
/** The interval `regressed` was decided on, in the dimension's native units. */
|
|
1076
|
+
ci: {
|
|
1077
|
+
low: number;
|
|
1078
|
+
high: number;
|
|
1079
|
+
};
|
|
1080
|
+
/** Which estimator produced `ci`. */
|
|
1081
|
+
decisionStatistic: PairedDecisionStatistic;
|
|
1082
|
+
/** McNemar's exact evidence on a pass/fail dimension; null otherwise. */
|
|
1083
|
+
mcnemar: PairedMcNemarEvidence | null;
|
|
1084
|
+
/** `ci` has zero width — no evidence in either direction. */
|
|
1085
|
+
indeterminate: boolean;
|
|
1086
|
+
/** True iff the candidate may have regressed this dimension by more than
|
|
1087
|
+
* tolerance: the lower bound of the DECIDING interval on (candidate −
|
|
1088
|
+
* baseline) is below −tolerance, OR the exact small-sample test proves a drop
|
|
1089
|
+
* past tolerance. */
|
|
1090
|
+
regressed: boolean;
|
|
1091
|
+
tolerance: number;
|
|
1092
|
+
n: number;
|
|
1093
|
+
}
|
|
1094
|
+
/** Detect the native scale of a set of scores: 0-100 when any magnitude clears
|
|
1095
|
+
* 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
|
|
1096
|
+
* expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
|
|
1097
|
+
declare function detectScale(values: number[]): 1 | 100;
|
|
1098
|
+
/** Per-critical-dimension regression guard. For each dimension, pair the
|
|
1099
|
+
* candidate vs baseline values by full cellId and bootstrap the paired delta;
|
|
1100
|
+
* a dimension is "regressed" when the CI lower bound < −tolerance (conservative
|
|
1101
|
+
* — blocks if the credible worst case exceeds tolerance, which is the right
|
|
1102
|
+
* posture for safety dimensions like `hallucination_free`). When `tolerance`
|
|
1103
|
+
* is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
|
|
1104
|
+
*
|
|
1105
|
+
* The interval comes from {@link decidePairedPromotion}, so a pass/fail
|
|
1106
|
+
* dimension is judged on Tango's score interval rather than a percentile
|
|
1107
|
+
* bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
|
|
1108
|
+
* is not a valid interval at one. That matters most here because this guard
|
|
1109
|
+
* fails OPEN by construction: `tolerance` is positive, so an interval pinned at
|
|
1110
|
+
* [0,0] never satisfies `low < −tolerance` and a real regression on a safety
|
|
1111
|
+
* dimension would be reported as `regressed: false`. On the median it fails the
|
|
1112
|
+
* same way for the same reason — when most pairs tie, which is automatic for a
|
|
1113
|
+
* pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
|
|
1114
|
+
* to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
|
|
1115
|
+
* restore the pre-0.134 behaviour. */
|
|
1116
|
+
declare function dimensionRegressions(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, criticalDimensions: string[], opts?: {
|
|
1117
|
+
tolerance?: number;
|
|
1118
|
+
confidence?: number;
|
|
1119
|
+
resamples?: number;
|
|
1120
|
+
seed?: number;
|
|
1121
|
+
/** Paired statistic the CI is computed on. Default `'mean'` — see
|
|
1122
|
+
* {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */
|
|
1123
|
+
statistic?: 'mean' | 'median';
|
|
1124
|
+
}): DimensionRegression[];
|
|
1125
|
+
//#endregion
|
|
1126
|
+
export { paretoSignificanceGate as $, PairArmsOptions as A, PowerPreflight as B, hashJson as C, ComparePairedArmsOptions as D, verifyManifest as E, PairedCorrectness as F, BuildEvidenceVectorOptions as G, powerPreflight as H, PairedMetricDelta as I, ParetoSignificanceGateOptions as J, EvidenceVector as K, comparePairedArms as L, PairRunRecordsResult as M, PairedArmRow as N, MatchedPair as O, PairedArmsComparison as P, paretoPolicy as Q, pairArms as R, evaluateHypothesis as S, signManifest as T, AxisEvidence as U, PowerPreflightOptions as V, AxisVerdict as W, PromotionPolicy as X, PromotionObjective as Y, buildEvidenceVector as Z, sequentialPairedGate as _, pairedRiskDifference as _t, detectScale as a, EProcessOptions as at, SignedManifest as b, passAtK as bt, pairHoldout as c, eProcess as ct, SequentialDecision as d, ProportionInterval as dt, Objective as et, SequentialObservation as f, RiskDifferenceResult as ft, sequentialDecide as g, pairedBinaryScale as gt, SequentialStreamState as h, mcnemar as ht, PairedHoldout as i, EProcess as it, PairArmsResult as j, MatchedRunRecordPair as k, SequentialDecideFn as l, ExactRiskDifferenceResult as lt, SequentialPairedGateOptions as m, isBinaryOutcomeVector as mt, HeldoutSignificance as n, dominates as nt, dimensionRegressions as o, EProcessState as ot, SequentialPairedGate as p, ScoreRiskDifferenceResult as pt, ObjectiveSource as q, HeldoutSignificanceOptions as r, paretoFrontier as rt, heldoutSignificance as s, EProcessStep as st, DimensionRegression as t, ParetoResult as tt, SequentialDecideOptions as u, McNemarResult as ut, HypothesisManifest as v, pairedRiskDifferenceExact as vt, manifestContentDigest as w, SignedManifestAlgo as x, wilson as xt, HypothesisResult as y, pairedRiskDifferenceScore as yt, pairRunRecords as z };
|
|
1127
|
+
//# sourceMappingURL=statistical-heldout-Cqb73yE9.d.ts.map
|