@tangle-network/agent-eval 0.173.3 → 0.175.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (149) hide show
  1. package/CHANGELOG.md +54 -0
  2. package/README.md +1 -1
  3. package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
  4. package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
  5. package/dist/adapters/http.d.ts +2 -2
  6. package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
  7. package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
  8. package/dist/analyst/index.d.ts +7 -9
  9. package/dist/analyst/index.d.ts.map +1 -1
  10. package/dist/analyst/index.js +8 -8
  11. package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
  12. package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
  13. package/dist/{benchmark-command-BY9oscke.js → benchmark-command-D_5xG9LG.js} +13 -13
  14. package/dist/{benchmark-command-BY9oscke.js.map → benchmark-command-D_5xG9LG.js.map} +1 -1
  15. package/dist/benchmarks/index.d.ts +3 -4
  16. package/dist/benchmarks/index.d.ts.map +1 -1
  17. package/dist/benchmarks/index.js +3 -3
  18. package/dist/campaign/index.d.ts +5 -9
  19. package/dist/campaign/index.js +7 -7
  20. package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
  21. package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
  22. package/dist/{opencode-sqlite-eK6HW6dr.js → claude-jsonl-CxZZrDJ3.js} +9 -149
  23. package/dist/claude-jsonl-CxZZrDJ3.js.map +1 -0
  24. package/dist/cli.js +9 -2
  25. package/dist/cli.js.map +1 -1
  26. package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
  27. package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
  28. package/dist/contract/index.d.ts +9 -10
  29. package/dist/contract/index.js +8 -8
  30. package/dist/{default-registry-B0bKikCb.js → default-registry-DBqVI4pq.js} +5 -5
  31. package/dist/{default-registry-B0bKikCb.js.map → default-registry-DBqVI4pq.js.map} +1 -1
  32. package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
  33. package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
  34. package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
  35. package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
  36. package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
  37. package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
  38. package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
  39. package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
  40. package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
  41. package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
  42. package/dist/experiment/index.d.ts +1 -4
  43. package/dist/experiment/index.d.ts.map +1 -1
  44. package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
  45. package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
  46. package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
  47. package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
  48. package/dist/fuzz.js +1 -1
  49. package/dist/fuzz.js.map +1 -1
  50. package/dist/hosted/index.d.ts +1 -1
  51. package/dist/{index-D0Db5X-4.d.ts → index-BAAiSF3_.d.ts} +5 -5
  52. package/dist/{index-D0Db5X-4.d.ts.map → index-BAAiSF3_.d.ts.map} +1 -1
  53. package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
  54. package/dist/index-BTrx5s8m.d.ts.map +1 -0
  55. package/dist/index-DKXuBPXf.d.ts +3840 -0
  56. package/dist/index-DKXuBPXf.d.ts.map +1 -0
  57. package/dist/index.d.ts +11 -13
  58. package/dist/index.d.ts.map +1 -1
  59. package/dist/index.js +11 -11
  60. package/dist/{integrity-BWywb34E.js → integrity-DsHWCebQ.js} +11 -435
  61. package/dist/integrity-DsHWCebQ.js.map +1 -0
  62. package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
  63. package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
  64. package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
  65. package/dist/llm-judge-DmNaBrXB.js.map +1 -0
  66. package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
  67. package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
  68. package/dist/multishot/golden/index.d.ts +1 -1
  69. package/dist/multishot/index.d.ts +2 -2
  70. package/dist/openapi.json +1 -1
  71. package/dist/opencode-sqlite-CNw3vubS.js +145 -0
  72. package/dist/opencode-sqlite-CNw3vubS.js.map +1 -0
  73. package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
  74. package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
  75. package/dist/report-command-DKlXfU5r.js +1528 -0
  76. package/dist/report-command-DKlXfU5r.js.map +1 -0
  77. package/dist/rl.d.ts +1 -1
  78. package/dist/rl.d.ts.map +1 -1
  79. package/dist/rl.js.map +1 -1
  80. package/dist/rollout/index.js +3 -2
  81. package/dist/{rollout-C-znbbYg.js → rollout-CGlDq1GI.js} +3 -2
  82. package/dist/{rollout-C-znbbYg.js.map → rollout-CGlDq1GI.js.map} +1 -1
  83. package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
  84. package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
  85. package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
  86. package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
  87. package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
  88. package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
  89. package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
  90. package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
  91. package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
  92. package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
  93. package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
  94. package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
  95. package/dist/supervisor-run/index.d.ts +71 -6
  96. package/dist/supervisor-run/index.d.ts.map +1 -1
  97. package/dist/supervisor-run/index.js +6 -1357
  98. package/dist/supervisor-run/index.js.map +1 -1
  99. package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
  100. package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
  101. package/dist/terminal-record-Ce9_UjRz.js +539 -0
  102. package/dist/terminal-record-Ce9_UjRz.js.map +1 -0
  103. package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
  104. package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
  105. package/dist/trace-repair/index.d.ts +1 -1
  106. package/dist/traces.d.ts +2 -2
  107. package/dist/traces.js +4 -4
  108. package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
  109. package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
  110. package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
  111. package/dist/types-DQ0e2E7y.js.map +1 -0
  112. package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
  113. package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
  114. package/dist/{types-CoPUTiXb.d.ts → types-vUdAx2Cj.d.ts} +65 -3
  115. package/dist/types-vUdAx2Cj.d.ts.map +1 -0
  116. package/docs/campaign-proposers.md +42 -0
  117. package/package.json +1 -1
  118. package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
  119. package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
  120. package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
  121. package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
  122. package/dist/benchmark-BjLGkfnN.d.ts +0 -236
  123. package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
  124. package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
  125. package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
  126. package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
  127. package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
  128. package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
  129. package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
  130. package/dist/index-BQqOjerE.d.ts.map +0 -1
  131. package/dist/index-CFDffsKz.d.ts +0 -1135
  132. package/dist/index-CFDffsKz.d.ts.map +0 -1
  133. package/dist/integrity-BWywb34E.js.map +0 -1
  134. package/dist/llm-judge-BfqMFo4h.js.map +0 -1
  135. package/dist/opencode-sqlite-eK6HW6dr.js.map +0 -1
  136. package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
  137. package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
  138. package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
  139. package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
  140. package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
  141. package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
  142. package/dist/proposal-findings-bko3GGy-.js.map +0 -1
  143. package/dist/provenance-CRY67X50.d.ts +0 -1995
  144. package/dist/provenance-CRY67X50.d.ts.map +0 -1
  145. package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
  146. package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
  147. package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
  148. package/dist/types-CiWITkGo.js.map +0 -1
  149. package/dist/types-CoPUTiXb.d.ts.map +0 -1
@@ -0,0 +1,1127 @@
1
+ import { a as RunRecord } from "./run-record-DTv1MdjK.js";
2
+ import { a as PairedPromotionDecision, d as PairedBootstrapResult, i as PairedMcNemarEvidence, r as PairedDecisionStatistic, t as PairedDecisionMethod, u as PairedBootstrapOptions } from "./paired-promotion-decision-CGzg0cI_.js";
3
+ import { R as Scenario, b as GenerationRecord, h as GateContext, p as Gate, v as GateResult, w as JudgeScore } from "./types-BJz2CPTM.js";
4
+ //#region src/statistics/paired-binary.d.ts
5
+ /** A binomial proportion estimate with a confidence interval. */
6
+ interface ProportionInterval {
7
+ /** Point estimate successes / n (0 when n = 0). */
8
+ estimate: number;
9
+ /** Lower bound, clamped to [0, 1]. */
10
+ lower: number;
11
+ /** Upper bound, clamped to [0, 1]. */
12
+ upper: number;
13
+ }
14
+ /**
15
+ * Wilson score interval for a binomial proportion. Correct at small n and near
16
+ * 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
17
+ * understates coverage. Use this for any pass-rate / hit-rate / realness-rate
18
+ * CI — the continuous `confidenceInterval` assumes the wrong distribution for a
19
+ * proportion. `n = 0 ⇒ {0, 0, 0}`.
20
+ */
21
+ declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
22
+ /**
23
+ * Are these per-item outcomes binary (every value exactly 0 or 1)?
24
+ *
25
+ * The discriminator a promotion gate needs before choosing a paired statistic.
26
+ * On binary outcomes the paired delta vector lives in {-1, 0, +1} and is
27
+ * normally dominated by zeros (both arms solve, or both arms miss, most items),
28
+ * so its MEDIAN is pinned at exactly 0 no matter how large the real shift in
29
+ * success rate is — and a bootstrap CI on that median collapses to [0, 0].
30
+ * A gate keying on `ci.low > threshold` is then structurally unable to see
31
+ * either a gain or a regression. Detect this shape and switch to the
32
+ * paired-binary estimators ({@link mcnemar}, {@link pairedRiskDifference})
33
+ * instead of silently answering "no" forever.
34
+ *
35
+ * Empty input is NOT binary: there is no evidence of the outcome's shape, and
36
+ * defaulting an empty vector into the binary branch would pick a statistic on
37
+ * no data at all.
38
+ *
39
+ * NOT the right discriminator for a gate. It recognises the literal {0, 1}
40
+ * encoding and nothing else, so a pass/fail dimension emitted on 0-100 — which
41
+ * judges in this codebase do routinely — reads as non-binary, and a single
42
+ * partial-credit score in an otherwise pass/fail vector flips it to false while
43
+ * leaving the median just as blind. Gates want {@link pairedBinaryScale} (any
44
+ * two-point encoding). This predicate remains for callers that specifically
45
+ * mean "literally 0/1".
46
+ */
47
+ declare function isBinaryOutcomeVector(values: ArrayLike<number>): boolean;
48
+ /** Result of a McNemar paired-binary significance test. */
49
+ interface McNemarResult {
50
+ /** Total paired observations. */
51
+ n: number;
52
+ /** Discordant pairs (b + c) — the only ones that carry signal. */
53
+ nDiscordant: number;
54
+ /** Pairs where treatment succeeded and control failed ("newly correct"). */
55
+ b: number;
56
+ /** Pairs where control succeeded and treatment failed ("newly wrong"). */
57
+ c: number;
58
+ /** Continuity-corrected chi-square statistic (reference; exact p drives the call). */
59
+ statistic: number;
60
+ /** Two-sided p-value. Exact (binomial sign test on discordant pairs). */
61
+ pValue: number;
62
+ }
63
+ /**
64
+ * McNemar's test for paired binary outcomes — the correct significance test for
65
+ * "does treatment change the success rate vs control on the SAME items". Only
66
+ * discordant pairs (one arm right, the other wrong) carry information; concordant
67
+ * pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
68
+ * rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
69
+ * Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
70
+ * at the small discordant counts typical of eval runs (no continuity-corrected
71
+ * chi-square approximation needed, though it is returned as `statistic` for
72
+ * reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
73
+ * the module's (before, after) convention. Throws on unequal lengths.
74
+ */
75
+ declare function mcnemar(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>): McNemarResult;
76
+ /** A paired binary effect size (treatment rate − control rate) with a CI. */
77
+ interface RiskDifferenceResult {
78
+ /** Total paired observations. */
79
+ n: number;
80
+ /** Discordant pairs: treatment-win count. */
81
+ b: number;
82
+ /** Discordant pairs: control-win count. */
83
+ c: number;
84
+ /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
85
+ riskDifference: number;
86
+ /** Lower bound of the CI, clamped to [-1, 1]. */
87
+ lower: number;
88
+ /** Upper bound of the CI, clamped to [-1, 1]. */
89
+ upper: number;
90
+ /** Confidence level used. */
91
+ confidence: number;
92
+ }
93
+ /**
94
+ * Paired risk difference (the effect-size companion to {@link mcnemar}): the
95
+ * change in success rate p(treatment) − p(control) on matched items, which for
96
+ * paired binary data equals (b − c) / n. The CI uses the paired variance from
97
+ * the discordant counts, not the independent-samples formula (which overstates
98
+ * the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
99
+ * arrays, control first. Throws on unequal lengths.
100
+ *
101
+ * REPORTING ONLY — do NOT decide a promotion on this interval. The CI is a Wald
102
+ * normal approximation, which badly UNDERCOVERS when only a handful of pairs are
103
+ * discordant: at n = 3 with b = 2, c = 0 it returns [0.133, 1.000], excluding 0,
104
+ * while McNemar's exact test on the same data gives p = 0.50. A gate keying on
105
+ * `lower > 0` would promote noise. Use {@link pairedRiskDifferenceExact}, whose
106
+ * interval is dual to the exact test by construction, for any decision.
107
+ */
108
+ declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
109
+ /** A paired binary effect size with an EXACT interval and the exact test that
110
+ * bounds it — one object so a caller cannot read the estimate without the
111
+ * significance it is entitled to. */
112
+ interface ExactRiskDifferenceResult {
113
+ /** Total paired observations. */
114
+ n: number;
115
+ /** Discordant pairs: treatment-win count. */
116
+ b: number;
117
+ /** Discordant pairs: control-win count. */
118
+ c: number;
119
+ /** Discordant pairs (b + c) — the only ones carrying information. */
120
+ nDiscordant: number;
121
+ /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
122
+ riskDifference: number;
123
+ /** Exact conditional CI lower bound. 0 when there are no discordant pairs. */
124
+ lower: number;
125
+ /** Exact conditional CI upper bound. 0 when there are no discordant pairs. */
126
+ upper: number;
127
+ /** Confidence level used. */
128
+ confidence: number;
129
+ /** McNemar's exact two-sided p-value on the same discordant counts. */
130
+ pValue: number;
131
+ }
132
+ /**
133
+ * Paired risk difference with the EXACT CONDITIONAL interval — the estimator a
134
+ * promotion gate may decide on.
135
+ *
136
+ * Conditional on the number of discordant pairs m = b + c, the treatment-win
137
+ * count b is Binomial(m, π) with π = P(treatment wins | discordant), and the
138
+ * risk difference is an exact reparameterisation: RD = (2π − 1)·m/n. So a
139
+ * Clopper-Pearson exact interval for π maps straight onto RD. This buys the
140
+ * property the Wald interval in {@link pairedRiskDifference} does not have:
141
+ *
142
+ * **`lower > 0` ⟺ McNemar's exact test rejects at α = 1 − confidence.**
143
+ *
144
+ * Clopper-Pearson excludes π = 0.5 exactly when the two-sided exact binomial
145
+ * test of π = 0.5 rejects, and that test IS {@link mcnemar}'s p-value — so the
146
+ * interval and the test can never disagree, and a gate keyed on `lower` cannot
147
+ * promote what the exact test refuses. The exact p is returned in the same
148
+ * object so the two are impossible to compute apart.
149
+ *
150
+ * The interval is conservative (exact intervals over-cover; conditioning on m
151
+ * discards the concordant pairs' information about m itself). That is the
152
+ * correct direction for a promotion gate: it refuses more often, never less.
153
+ *
154
+ * With m = 0 there are no discordant pairs and π is not identified: the result
155
+ * is the degenerate [0, 0] with p = 1. That is NOT evidence of equivalence —
156
+ * callers must treat a zero-width interval as "cannot decide", not as "no
157
+ * difference". Inputs are paired 0/1 (or boolean) arrays, control first.
158
+ * Throws on unequal lengths.
159
+ */
160
+ declare function pairedRiskDifferenceExact(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ExactRiskDifferenceResult;
161
+ /** A paired binary effect size with an interval that is valid at a NONZERO
162
+ * margin — the estimator a noninferiority decision may be made on. */
163
+ interface ScoreRiskDifferenceResult {
164
+ /** Total paired observations. */
165
+ n: number;
166
+ /** Discordant pairs: treatment-win count. */
167
+ b: number;
168
+ /** Discordant pairs: control-win count. */
169
+ c: number;
170
+ /** Discordant pairs (b + c). */
171
+ nDiscordant: number;
172
+ /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
173
+ riskDifference: number;
174
+ /** Score-interval lower bound on the population risk difference. */
175
+ lower: number;
176
+ /** Score-interval upper bound on the population risk difference. */
177
+ upper: number;
178
+ /** Confidence level used. */
179
+ confidence: number;
180
+ }
181
+ /**
182
+ * Paired risk difference with TANGO'S (1998) SCORE INTERVAL — the estimator a
183
+ * promotion gate may decide on **at a nonzero margin**.
184
+ *
185
+ * {@link pairedRiskDifferenceExact} conditions on the observed discordant count
186
+ * `m = b + c`, builds a Clopper-Pearson interval for the win share among those
187
+ * `m` pairs, and multiplies by the observed `m/n`. That is exact for testing
188
+ * RD = 0 — it is dual to McNemar — but it is NOT a confidence interval for the
189
+ * population risk difference at a nonzero margin, because the sampling
190
+ * variability of `m/n` itself is discarded. The gap is not academic: with the
191
+ * production caller's `pairedDeltaThreshold: -0.05`, a process whose true risk
192
+ * difference sits exactly on that margin clears a nominal-95 % `lower > margin`
193
+ * check 24.75 % of the time at n = 40 and 43.95 % at n = 76 (2000 replicates
194
+ * each) when the conditional interval decides.
195
+ *
196
+ * Tango's interval inverts the score test of RD = delta, which estimates the
197
+ * nuisance loss rate under each hypothesised delta instead of fixing it at the
198
+ * observed value, so `m` contributes its own uncertainty. It is the method
199
+ * `ratesci::scorepairci` uses for paired risk-difference noninferiority, and it
200
+ * is not conditional, so it stays valid as the margin moves away from zero.
201
+ *
202
+ * The bounds are found by bisecting `tangoScore(delta) = ±z` — the score is
203
+ * monotone decreasing in delta, so each crossing is unique. Inputs are paired
204
+ * 0/1 (or boolean) arrays, control first. Throws on unequal lengths.
205
+ */
206
+ declare function pairedRiskDifferenceScore(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ScoreRiskDifferenceResult;
207
+ /**
208
+ * The common positive level `s` such that EVERY value across both paired arms is
209
+ * exactly 0 or `s` — i.e. the outcome is pass/fail, whatever encoding it arrived
210
+ * in. Returns null when the outcomes are not two-point, when the two arms use
211
+ * different levels, or when no positive value was observed at all (all-zero
212
+ * arms: the level is not identified, and there is nothing to decide anyway).
213
+ *
214
+ * This is the scale-aware successor to {@link isBinaryOutcomeVector}, which only
215
+ * recognises literal {0, 1}. Judges in this codebase emit dimensions on 0-100 as
216
+ * well as [0,1] (see `detectScale` in `campaign/gates/statistical-heldout.ts`),
217
+ * so a pass/fail dimension routinely arrives as {0, 100} and a {0,1}-only test
218
+ * silently sends it down the median path that cannot see it. Any positive level
219
+ * is accepted, not just 1 and 100: for a two-point {0, s} outcome the mean paired
220
+ * delta is exactly s·(b − c)/n, so the binary estimators apply after dividing by
221
+ * s and rescaling the result back into the caller's native units.
222
+ *
223
+ * Non-finite values ⇒ null: an unusable outcome must not be classified as a
224
+ * clean pass/fail shape.
225
+ */
226
+ declare function pairedBinaryScale(before: ArrayLike<number>, after: ArrayLike<number>): number | null;
227
+ /**
228
+ * Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
229
+ * Language Models Trained on Code"). Given `n` independent samples for one
230
+ * problem of which `c` pass, the probability that at least one of a random k of
231
+ * them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
232
+ * first k pass" is biased high at small n; this is the variance-reduced estimator
233
+ * averaged implicitly over all k-subsets. Average the per-problem values across
234
+ * the suite for the corpus pass@k. Computed in the numerically stable product
235
+ * form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
236
+ */
237
+ declare function passAtK(n: number, c: number, k: number): number;
238
+ //#endregion
239
+ //#region src/statistics/sequential-eprocess.d.ts
240
+ interface EProcessOptions {
241
+ /** Type-I error budget. The process decides when wealth ≥ 1/alpha
242
+ * (Ville's inequality). Default 0.05. */
243
+ alpha?: number;
244
+ /** Truncation bound on the predictable bet λ ∈ [0, maxBet]. Must satisfy
245
+ * maxBet < 1/nullMean so every wealth factor stays strictly positive.
246
+ * Default 0.5. */
247
+ maxBet?: number;
248
+ /** The null boundary m₀ for H0: E[x] ≤ m₀ on x ∈ [0,1]. Default 0.5
249
+ * (the paired-delta encoding x = (d+1)/2 maps "no effect" to 1/2).
250
+ * A pre-registered minEffect shifts this — see `sequentialPairedGate`. */
251
+ nullMean?: number;
252
+ /** Continue a process from a `state()` snapshot taken before a restart.
253
+ * The snapshot never supplies the parameters: `alpha`, `maxBet`, and
254
+ * `nullMean` resolve from this options object exactly as for a fresh
255
+ * process, and a snapshot recorded under different parameters is refused
256
+ * (`ValidationError`) — re-deciding the same stream under new parameters
257
+ * would reopen optional stopping. A snapshot whose fields are not
258
+ * mutually consistent (tampered or truncated) is refused the same way. */
259
+ resume?: EProcessState;
260
+ }
261
+ interface EProcessStep {
262
+ /** Current wealth W_n — the e-value against H0 after n observations. */
263
+ wealth: number;
264
+ /** Observations consumed so far. */
265
+ n: number;
266
+ /** True from the first n where W_n ≥ 1/alpha onward (sticky). */
267
+ decided: boolean;
268
+ }
269
+ /**
270
+ * Complete snapshot of an e-process. Together with the parameters it is
271
+ * sufficient to continue the process after a restart: `sumX` and `varSum`
272
+ * are the running sums the next bet is computed from, so a process rebuilt
273
+ * from a snapshot produces the same wealth sequence and decision as one that
274
+ * was never interrupted. Plain data: a JSON round-trip preserves it
275
+ * (`decidedAtN` is undefined, and therefore omitted, until decided).
276
+ */
277
+ interface EProcessState extends EProcessStep {
278
+ alpha: number;
279
+ maxBet: number;
280
+ nullMean: number;
281
+ /** The decision boundary 1/alpha. */
282
+ threshold: number;
283
+ /** Observation count at the first threshold crossing; undefined until decided. */
284
+ decidedAtN?: number;
285
+ /** Σ x_i over the n observations consumed. */
286
+ sumX: number;
287
+ /** Σ (x_i − μ̂_i)² over the n observations consumed, μ̂_i the shrunk running
288
+ * mean after observation i. */
289
+ varSum: number;
290
+ }
291
+ interface EProcess {
292
+ /** Consume one observation x ∈ [0,1]. Throws on non-finite / out-of-range
293
+ * input — a silent clamp would corrupt the type-I guarantee. */
294
+ update(x: number): EProcessStep;
295
+ state(): EProcessState;
296
+ }
297
+ /**
298
+ * Betting test-martingale for bounded observations — the e-process core of
299
+ * anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
300
+ * of bounded random variables by betting", JRSS-B 2024).
301
+ *
302
+ * Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
303
+ *
304
+ * W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
305
+ *
306
+ * with the truncated GROW-style plug-in bet computed from PRIOR observations:
307
+ *
308
+ * λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
309
+ *
310
+ * where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
311
+ * σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
312
+ *
313
+ * PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
314
+ * ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
315
+ * E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
316
+ * supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
317
+ * type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
318
+ * (no prior evidence), so the first observation never moves wealth.
319
+ *
320
+ * `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
321
+ * wealth keeps updating after the crossing (the e-process remains valid), but
322
+ * the decision time is the first crossing.
323
+ */
324
+ declare function eProcess(opts?: EProcessOptions): EProcess;
325
+ //#endregion
326
+ //#region src/pareto.d.ts
327
+ /**
328
+ * Pareto frontier — multi-objective optimization over candidate runs.
329
+ *
330
+ * Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
331
+ * trading off (cost, latency, quality) or (passRate, tokenBudget,
332
+ * ttfb), you rarely have a single "winner" — you have a set of
333
+ * non-dominated candidates. This module exposes:
334
+ *
335
+ * - `paretoFrontier`: filter a set of candidates to the non-dominated ones
336
+ * - `dominates`: does A dominate B across all objectives?
337
+ *
338
+ * Each objective is declared with a direction: 'maximize' (higher=better)
339
+ * or 'minimize' (lower=better). Candidates are any object; pass an
340
+ * `objective(candidate)` accessor.
341
+ */
342
+ type Direction = 'maximize' | 'minimize';
343
+ interface Objective<T> {
344
+ /** Stable label used in reports. */
345
+ name: string;
346
+ direction: Direction;
347
+ value: (candidate: T) => number;
348
+ }
349
+ interface ParetoResult<T> {
350
+ frontier: T[];
351
+ dominated: T[];
352
+ /** Index map: frontier[i] dominates each of dominatedBy[i]. */
353
+ dominanceMap: Array<{
354
+ dominator: T;
355
+ dominated: T[];
356
+ }>;
357
+ }
358
+ /** Does candidate A weakly dominate B — strictly better on at least one objective and no worse on any? */
359
+ declare function dominates<T>(a: T, b: T, objectives: Objective<T>[]): boolean;
360
+ /**
361
+ * Compute the non-dominated frontier. Candidates with NaN/Infinity on any
362
+ * objective are excluded (can't rank them). A candidate enters the frontier
363
+ * iff no other candidate dominates it.
364
+ */
365
+ declare function paretoFrontier<T>(candidates: T[], objectives: Objective<T>[]): ParetoResult<T>;
366
+ //#endregion
367
+ //#region src/campaign/gates/promotion-policy.d.ts
368
+ /** Where an objective's per-cell scalar comes from. `composite` reads the
369
+ * judge's composite; `dimension` reads a named per-dimension score. */
370
+ type ObjectiveSource = {
371
+ kind: 'composite';
372
+ } | {
373
+ kind: 'dimension';
374
+ dimension: string;
375
+ };
376
+ interface PromotionObjective {
377
+ /** Stable label used in reports + `contributingGates`. */
378
+ name: string;
379
+ source: ObjectiveSource;
380
+ /** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients
381
+ * the paired delta so a positive bootstrap always means "candidate better". */
382
+ direction: Direction;
383
+ /** The good-direction paired-delta CI lower bound must EXCEED this to count
384
+ * as a significant gain on this axis. Interpreted in the judge's native
385
+ * scale. Default 0 (⇒ "confidently better"). */
386
+ gainThreshold?: number;
387
+ /** A floor breach (regression) is declared when the good-direction CI lower
388
+ * bound is below −floorTolerance, or when the exact small-sample test proves
389
+ * a drop past it. When omitted it auto-scales off observed magnitudes
390
+ * (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */
391
+ floorTolerance?: number;
392
+ }
393
+ /** Per-axis verdict from the good-direction paired bootstrap. */
394
+ type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs';
395
+ interface AxisEvidence {
396
+ name: string;
397
+ source: ObjectiveSource;
398
+ direction: Direction;
399
+ /** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):
400
+ * a positive value means the candidate is better on this axis.
401
+ *
402
+ * DIAGNOSTIC on a pass/fail axis: there the verdict is decided on Tango's
403
+ * score interval instead, because a percentile bootstrap over a three-atom
404
+ * delta lattice is not a valid interval at the nonzero margin `floorTolerance`
405
+ * and `gainThreshold` create. `ci` carries the interval that decided. */
406
+ bootstrap: PairedBootstrapResult;
407
+ /** Which paired statistic `bootstrap.low`/`.high` bracket. `'mean'` unless the
408
+ * caller asked for the median — on a pass/fail axis the median and its whole
409
+ * CI are pinned at 0 by tie domination and can see neither a gain nor a
410
+ * regression. `bootstrap.median` still carries the median point estimate. */
411
+ bootstrapStatistic: 'median' | 'mean';
412
+ /** The interval the axis verdict was actually decided on, good-direction and
413
+ * in the axis's native units. */
414
+ ci: {
415
+ low: number;
416
+ high: number;
417
+ };
418
+ /** Which estimator produced `ci`. */
419
+ decisionStatistic: PairedDecisionStatistic;
420
+ /** McNemar's exact evidence on a pass/fail axis; null otherwise. */
421
+ mcnemar: PairedMcNemarEvidence | null;
422
+ /** `ci` has zero width — no evidence in either direction, so the axis is
423
+ * neither improved nor regressed however the point estimate sits. */
424
+ indeterminate: boolean;
425
+ /** Paired observations contributing to this axis. */
426
+ n: number;
427
+ minimumRequired: number;
428
+ decisionMethod: PairedDecisionMethod;
429
+ gainThreshold: number;
430
+ floorTolerance: number;
431
+ verdict: AxisVerdict;
432
+ }
433
+ interface EvidenceVector {
434
+ /** One entry per objective — NOTHING averaged across axes. */
435
+ axes: AxisEvidence[];
436
+ /** Smallest paired n across axes that produced observations — the binding
437
+ * evidence-sufficiency constraint. 0 when no axis produced observations. */
438
+ minN: number;
439
+ /** Aggregate per-side cost from the gate context (a constraint input, not a
440
+ * CI axis — see the module header). */
441
+ cost: {
442
+ candidate: number;
443
+ baseline: number;
444
+ };
445
+ }
446
+ /** A promotion strategy: a pure function from the evidence vector to a verdict.
447
+ * Many policies can run over the same `EvidenceVector` and disagree — that's
448
+ * the point (competing strategies, shared evidence). */
449
+ type PromotionPolicy = (ev: EvidenceVector) => GateResult;
450
+ interface BuildEvidenceVectorOptions {
451
+ /** Minimum paired observations before an axis can claim significance; below
452
+ * it the axis is `few_runs`. The exact small-sample test may require more
453
+ * observations at the selected confidence. */
454
+ minProductiveRuns?: number;
455
+ /** Confidence level for every axis bootstrap. Default 0.95. */
456
+ confidence?: number;
457
+ /** Bootstrap resamples. Default 2000. */
458
+ resamples?: number;
459
+ /** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */
460
+ seed?: number;
461
+ /** Paired statistic every axis CI is computed on. Default `'mean'` — see
462
+ * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */
463
+ statistic?: 'mean' | 'median';
464
+ }
465
+ /**
466
+ * The Evidence Bus. For each objective, pair candidate vs baseline by full
467
+ * cellId and bootstrap a CI on the good-direction paired delta. Reuses the
468
+ * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
469
+ * a single source of truth governs pairing granularity + scale handling.
470
+ */
471
+ declare function buildEvidenceVector<TArtifact, TScenario extends Scenario>(ctx: GateContext<TArtifact, TScenario>, objectives: PromotionObjective[], opts?: BuildEvidenceVectorOptions): EvidenceVector;
472
+ /**
473
+ * The default strategy: symmetric multi-objective Pareto significance. Ship iff
474
+ * the candidate weakly dominates the baseline at the confidence level — no axis
475
+ * credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
476
+ * (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
477
+ * need_more_work. Statistically equivalent → hold (never ship noise).
478
+ */
479
+ declare const paretoPolicy: PromotionPolicy;
480
+ interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
481
+ /** The objective vector. Every axis is both a gain source and a safety floor. */
482
+ objectives: PromotionObjective[];
483
+ /** Strategy applied to the evidence vector. Default `paretoPolicy`. Override
484
+ * to run a stricter/looser strategy over the SAME bus (competing policies). */
485
+ policy?: PromotionPolicy;
486
+ /** Override the gate name in reports. */
487
+ name?: string;
488
+ }
489
+ /**
490
+ * Wrap the bus + a policy as a `Gate`. Plugs into the existing
491
+ * `runImprovementLoop({ gate })` slot and composes via `composeGate`; default
492
+ * loop behavior is unchanged because consumers opt in by passing this gate.
493
+ */
494
+ declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: ParetoSignificanceGateOptions): Gate<TArtifact, TScenario>;
495
+ //#endregion
496
+ //#region src/campaign/gates/power-preflight.d.ts
497
+ /**
498
+ * Power preflight — "can this budget detect the effect you are hunting?"
499
+ *
500
+ * The failure it prevents (measured, twice): a live prompt-improvement campaign ran
501
+ * 333 sandbox cells over 5.6 hours and produced a +0.08 holdout lift the ship gate
502
+ * (paired bootstrap, CI.low > 0.05) could not distinguish from zero — because at
503
+ * that holdout size and worker variance the MINIMUM DETECTABLE lift was larger than
504
+ * any effect a prompt change plausibly produces. The budget was spent learning what
505
+ * a 30-second calculation on the baseline cells already knew. No eval framework we
506
+ * know of surfaces this; every underpowered improvement run everywhere ends in an
507
+ * uninformative "hold".
508
+ *
509
+ * Model: the ship rule is `CI.low(paired Δ) > deltaThreshold`. Approximating the
510
+ * bootstrap CI as normal, `CI.low ≈ effect − z·sd_Δ/√n`, so the smallest shippable
511
+ * true effect is `MDE = deltaThreshold + z·sd_Δ/√n`. The paired-delta SD is unknown
512
+ * before the candidate exists; we bound it by the zero-correlation case
513
+ * `sd_Δ ≤ √2·sd_baseline` — a CONSERVATIVE (upper) MDE, which is the correct
514
+ * direction for a warning. Pairing is per cell (`scenario:rep`), so reps multiply n.
515
+ *
516
+ * Standalone by design: feed it any baseline composites (a `gate:'none'` run, a
517
+ * live-proof table) BEFORE budgeting the real search; `selfImprove` also attaches
518
+ * it to every result and warns when the run was structurally unable to ship.
519
+ */
520
+ interface PowerPreflightOptions {
521
+ /** Per-cell baseline composites on the HOLDOUT scenarios (one per scenario:rep cell). */
522
+ baselineComposites: number[];
523
+ /** Paired observations the budgeted comparison will produce
524
+ * (holdout scenarios × reps). Defaults to `baselineComposites.length`. */
525
+ pairedN?: number;
526
+ /** The ship gate's effect-size threshold. Default 0.05 (defaultProductionGate). */
527
+ deltaThreshold?: number;
528
+ /** CI confidence the gate uses. Default 0.95. */
529
+ confidence?: number;
530
+ /** True when the holdout is scored by the SAME judge/scorer family as the gate
531
+ * (selfImprove's default composition — one judge scores everything). Under a
532
+ * shared channel, raising paired n reduces only the IDIOSYNCRATIC noise share;
533
+ * systematic judge bias is untouched, so the MDE here is a lower bound and the
534
+ * only full debiaser is an independent second scoring channel
535
+ * (recursive-self-improvement S1c, closed form in EXP-023 P0). Default false. */
536
+ sharedScorerChannel?: boolean;
537
+ }
538
+ interface PowerPreflight {
539
+ /** Paired observations the comparison will have. */
540
+ n: number;
541
+ /** Baseline per-cell composite standard deviation (the variance the effect must beat). */
542
+ sd: number;
543
+ /** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */
544
+ mde: number;
545
+ /** Baseline holdout composite mean. */
546
+ baselineMean: number;
547
+ /** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */
548
+ headroom: number;
549
+ /** True when even the largest achievable effect (headroom) is below the MDE —
550
+ * the run is structurally unable to ship regardless of proposal quality.
551
+ * Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */
552
+ underpowered: boolean;
553
+ /** True when composites look [0,1]-scaled; headroom/underpowered are only
554
+ * meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */
555
+ scaleAssumed: boolean;
556
+ deltaThreshold: number;
557
+ confidence: number;
558
+ /** Set when the holdout shares the gate's scoring channel: more cells cannot
559
+ * buy back systematic judge bias — treat the MDE as a lower bound. */
560
+ sharedChannelCaveat?: string;
561
+ /** One actionable sentence for humans and logs. */
562
+ recommendation: string;
563
+ }
564
+ /** Estimate the minimum detectable lift a paired-holdout improvement run can
565
+ * ship at a given budget, from the baseline holdout composites — call it BEFORE
566
+ * spending a search to learn whether the effect you are hunting is even
567
+ * observable at this holdout size and worker variance. */
568
+ declare function powerPreflight(opts: PowerPreflightOptions): PowerPreflight;
569
+ //#endregion
570
+ //#region src/paired-arms.d.ts
571
+ /** One arm observation of one work item. Structural on purpose: callers
572
+ * project their own record type (e.g. a `RunRecord`) into this shape. */
573
+ interface PairedArmRow {
574
+ /** Matching key — rows sharing a `pairKey` across both arms form pairs
575
+ * (typically the task/scenario/seed identity). */
576
+ pairKey: string;
577
+ /** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on
578
+ * every row of a `pairKey` that has more than one rep in either arm; reps
579
+ * then pair only on exact (`pairKey`, `repKey`) match, never on outcome
580
+ * content. Optional when each arm has at most one rep of the item. */
581
+ repKey?: string;
582
+ /** Arm label this row was produced under. */
583
+ arm: string;
584
+ /** Binary outcome; omit when the comparison has no pass/fail notion. */
585
+ pass?: boolean;
586
+ /** Named numeric measurements (score, cost, latency, …). */
587
+ metrics?: Record<string, number>;
588
+ }
589
+ interface PairArmsOptions {
590
+ /** Arm treated as the control side of every pair. */
591
+ baselineArm: string;
592
+ /** Arm treated as the treatment side of every pair. */
593
+ treatmentArm: string;
594
+ }
595
+ /** One matched (baseline, treatment) observation of the same work item. */
596
+ interface MatchedPair {
597
+ pairKey: string;
598
+ /** 0-based position of this pair within its `pairKey`, ordered by sorted
599
+ * `repKey` (always 0 for a single-rep item). The rep identity itself is on
600
+ * the rows (`baseline.repKey` / `treatment.repKey`). */
601
+ repIndex: number;
602
+ baseline: PairedArmRow;
603
+ treatment: PairedArmRow;
604
+ }
605
+ interface PairArmsResult {
606
+ /** Matched pairs, ordered by (`pairKey`, `repIndex`). */
607
+ pairs: MatchedPair[];
608
+ /** Baseline rows left without a treatment counterpart — reported, never
609
+ * silently dropped. */
610
+ unpairedBaseline: PairedArmRow[];
611
+ /** Treatment rows left without a baseline counterpart. */
612
+ unpairedTreatment: PairedArmRow[];
613
+ }
614
+ /**
615
+ * Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
616
+ *
617
+ * A `pairKey` with at most one row per arm pairs directly, no `repKey`
618
+ * needed. A `pairKey` with multiple reps in either arm requires `repKey` on
619
+ * every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
620
+ * match — pairing is keyed purely on row identity, never on outcome content
621
+ * (outcome-keyed matching deflates discordant counts and biases McNemar), and
622
+ * is therefore independent of input order. Reps whose `repKey` has no
623
+ * counterpart in the other arm, and items present in only one arm, land in
624
+ * the unpaired lists — reported, never truncated.
625
+ *
626
+ * Fail-loud: throws when either named arm has zero rows (an unknown arm
627
+ * name would otherwise read as "everything unpaired"), when the two arm
628
+ * names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
629
+ * when a (`pairKey`, arm) group repeats a `repKey` (the match would be
630
+ * ambiguous).
631
+ */
632
+ declare function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult;
633
+ /** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */
634
+ interface PairedCorrectness {
635
+ /** Discordant pairs where the treatment passed and the baseline failed. */
636
+ b10: number;
637
+ /** Discordant pairs where the baseline passed and the treatment failed. */
638
+ b01: number;
639
+ /** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */
640
+ mcnemar: McNemarResult;
641
+ /** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */
642
+ riskDifference: RiskDifferenceResult;
643
+ }
644
+ /** Paired delta summary for one named metric (delta = treatment − baseline). */
645
+ interface PairedMetricDelta {
646
+ name: string;
647
+ /** Pairs where BOTH sides carry a finite value for this metric. */
648
+ n: number;
649
+ /** Pairs where at least one side does not carry the metric. */
650
+ nMissing: number;
651
+ /** Median paired delta, or null when `n === 0`. */
652
+ medianDelta: number | null;
653
+ /** Mean paired delta, or null when `n === 0`. */
654
+ meanDelta: number | null;
655
+ /** Bootstrap CI on the paired delta (`pairedBootstrap`); null when
656
+ * `n === 0` — a zero-width [0, 0] interval on no data would read as a
657
+ * measured tight null. */
658
+ bootstrapCi: PairedBootstrapResult | null;
659
+ /** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */
660
+ wilcoxon: {
661
+ w: number;
662
+ p: number;
663
+ } | null;
664
+ }
665
+ interface ComparePairedArmsOptions extends PairArmsOptions {
666
+ /** Metrics to compare. Default: every metric name observed on any matched
667
+ * pair, sorted. A name that appears on no pair is still reported (with
668
+ * `n = 0`) so a misspelled metric is visible instead of vanishing. */
669
+ metricNames?: string[];
670
+ /** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */
671
+ bootstrap?: PairedBootstrapOptions;
672
+ }
673
+ interface PairedArmsComparison {
674
+ nPairs: number;
675
+ nUnpairedBaseline: number;
676
+ nUnpairedTreatment: number;
677
+ /** null when no matched pair carries `pass` on both sides — a pass/fail
678
+ * verdict over rows that never measured pass/fail would be fabricated. */
679
+ correctness: PairedCorrectness | null;
680
+ metricDeltas: PairedMetricDelta[];
681
+ }
682
+ /**
683
+ * Full matched-pair arm comparison: pair via {@link pairArms}, then compose
684
+ * the paired estimators from `statistics` over the matched pairs.
685
+ *
686
+ * Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
687
+ * is that subset's size); each metric uses only the pairs where both sides
688
+ * carry a finite value for it, with the remainder counted in `nMissing`.
689
+ * Deltas are treatment − baseline throughout.
690
+ *
691
+ * Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
692
+ * non-finite metric value — silently treating corrupt telemetry as "metric
693
+ * absent" would misreport it as missing coverage.
694
+ */
695
+ declare function comparePairedArms(rows: readonly PairedArmRow[], opts: ComparePairedArmsOptions): PairedArmsComparison;
696
+ interface MatchedRunRecordPair {
697
+ pairKey: string;
698
+ repKey: string;
699
+ baseline: RunRecord;
700
+ treatment: RunRecord;
701
+ }
702
+ interface PairRunRecordsResult {
703
+ pairs: MatchedRunRecordPair[];
704
+ unpairedBaseline: RunRecord[];
705
+ unpairedTreatment: RunRecord[];
706
+ }
707
+ /**
708
+ * Pair two RunRecord arms by the identity of the evaluated work:
709
+ * `(experimentId, scenarioId, seed)`.
710
+ *
711
+ * Falling back to array order, candidate id, or experiment id can compare
712
+ * different tasks and fabricate lift. Duplicate identities throw.
713
+ */
714
+ declare function pairRunRecords(baselineRuns: readonly RunRecord[], treatmentRuns: readonly RunRecord[]): PairRunRecordsResult;
715
+ //#endregion
716
+ //#region src/pre-registration.d.ts
717
+ /**
718
+ * Pre-registered hypotheses — declare what you're testing BEFORE the
719
+ * run, check it AFTER. Prevents p-hacking, optional stopping, and the
720
+ * "we ran until it looked good" failure mode.
721
+ *
722
+ * Manifest is a plain JSON-friendly object. Sign it with a content hash
723
+ * + timestamp; the registered record becomes immutable. Post-run,
724
+ * evaluate the manifest against observed results — the library refuses
725
+ * to let you re-interpret a different metric as the declared one.
726
+ *
727
+ * A signed manifest is a portable record: it is written once and verified
728
+ * later, possibly by a different release. `algo` names the digest scheme it
729
+ * was signed under, and verification selects the encoder by that field, so a
730
+ * manifest signed by an earlier release still verifies.
731
+ */
732
+ interface HypothesisManifest {
733
+ id: string;
734
+ /** Human prose — goes into the audit trail. */
735
+ hypothesis: string;
736
+ /** Metric the hypothesis claims to move. */
737
+ metric: string;
738
+ /** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
739
+ direction: 'increase' | 'decrease';
740
+ /** Minimum effect size to count (same units as the metric). */
741
+ minEffect: number;
742
+ /** Alpha threshold. */
743
+ alpha: number;
744
+ /** Target statistical power at which sample size was pre-computed. */
745
+ power: number;
746
+ /** Declared N per arm before running. */
747
+ preRegisteredN: number;
748
+ /** ISO8601 timestamp the manifest was registered. */
749
+ registeredAt: string;
750
+ /** Optional identifiers to tie into the trace corpus. */
751
+ baselineLabel?: string;
752
+ candidateLabel?: string;
753
+ }
754
+ /**
755
+ * Identifier for the hashing scheme used to produce `contentHash`.
756
+ *
757
+ * Both schemes are sha256 hex over the manifest with `contentHash` and `algo`
758
+ * stripped, and differ only in how that manifest is serialized:
759
+ *
760
+ * - `'sha256-rfc8785'` — RFC 8785 canonical JSON. What {@link signManifest}
761
+ * emits.
762
+ * - `'sha256-content'` — key-sorted `JSON.stringify`. Read-only: manifests
763
+ * signed by an earlier release carry it, or carry no `algo` at all, and
764
+ * {@link verifyManifest} still verifies them.
765
+ */
766
+ type SignedManifestAlgo = 'sha256-content' | 'sha256-rfc8785';
767
+ interface SignedManifest extends HypothesisManifest {
768
+ /** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
769
+ contentHash: string;
770
+ /**
771
+ * Algorithm string describing how `contentHash` was produced.
772
+ *
773
+ * Optional on the type so serialized manifests without it still parse,
774
+ * but ALWAYS populated by {@link signManifest}. Consumers that want to
775
+ * enforce a known algorithm should reject manifests where this field
776
+ * is missing or unrecognized.
777
+ */
778
+ algo?: SignedManifestAlgo;
779
+ }
780
+ interface HypothesisResult {
781
+ manifest: SignedManifest;
782
+ observedN: number;
783
+ observedEffect: number;
784
+ observedPValue: number;
785
+ /** True iff the observed effect hits the pre-declared direction with
786
+ * magnitude ≥ minEffect AND p < alpha. */
787
+ confirmed: boolean;
788
+ /** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
789
+ rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
790
+ notes?: string;
791
+ }
792
+ /**
793
+ * SHA-256 hex (full 64 chars) over the RFC 8785 canonical JSON encoding of
794
+ * `obj` — the package's one identity scheme, shared with `ledger-core`.
795
+ *
796
+ * Values canonical JSON cannot represent faithfully — `undefined`, `NaN`,
797
+ * class instances, cycles — are refused rather than coerced, because a
798
+ * coercion maps two distinct records onto one digest.
799
+ *
800
+ * Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
801
+ * which takes a string input and returns a truncated 12-char prompt id.
802
+ *
803
+ * @example
804
+ * const hash = await hashJson({ id: '1', kind: 'spec' })
805
+ * // 'a3f1...' (64 hex chars)
806
+ */
807
+ declare function hashJson<T>(obj: T): Promise<string>;
808
+ /**
809
+ * Digest of a manifest under its own declared scheme, with `contentHash` and
810
+ * `algo` stripped. Synchronous, so a caller that must fail before consuming an
811
+ * observation does not have to await. Throws on an `algo` this release does
812
+ * not know — an unverifiable manifest must not read as a valid one.
813
+ */
814
+ declare function manifestContentDigest(manifest: SignedManifest): string;
815
+ /**
816
+ * Sign a manifest with a SHA-256 content hash over its RFC 8785 canonical
817
+ * JSON, with `contentHash` and `algo` stripped, and stamp the scheme in
818
+ * `algo` so a later reader knows which encoder to verify with.
819
+ */
820
+ declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
821
+ /**
822
+ * Verify that a signed manifest has not been tampered with, under the scheme
823
+ * the manifest itself declares.
824
+ */
825
+ declare function verifyManifest(m: SignedManifest): Promise<boolean>;
826
+ /**
827
+ * Evaluate a pre-registered hypothesis against observed results.
828
+ * Mechanical — no re-interpretation permitted.
829
+ */
830
+ declare function evaluateHypothesis(manifest: SignedManifest, observed: {
831
+ n: number;
832
+ effect: number;
833
+ pValue: number;
834
+ }): Promise<HypothesisResult>;
835
+ //#endregion
836
+ //#region src/campaign/gates/sequential.d.ts
837
+ type SequentialDecision = 'promote' | 'continue' | 'undecided-at-maxN';
838
+ interface SequentialObservation {
839
+ decision: SequentialDecision;
840
+ /** Current e-value (the betting wealth) against H0. */
841
+ eValue: number;
842
+ /** Paired deltas consumed so far. */
843
+ n: number;
844
+ /** Names the decision basis. For 'undecided-at-maxN' it states explicitly
845
+ * that exhausting the budget is NOT evidence of no effect. */
846
+ reason: string;
847
+ }
848
+ interface SequentialPairedGateOptions {
849
+ /** Type-I budget. With `preRegistration` bound this MUST match
850
+ * `manifest.alpha` (conflict throws). Default 0.05. */
851
+ alpha?: number;
852
+ /** Minimum paired deltas before a promote may fire. The stopping rule is
853
+ * "first n ≥ minN with e-value ≥ 1/alpha" — still a valid stopping time.
854
+ * Default 5. */
855
+ minN?: number;
856
+ /** Pre-registered observation budget. Required unless `preRegistration`
857
+ * supplies it via `preRegisteredN` (conflict throws). */
858
+ maxN?: number;
859
+ /** Bet truncation forwarded to `eProcess`. Default 0.5. */
860
+ maxBet?: number;
861
+ /** Bound on |delta| in the judge's native scale; deltas are mapped to
862
+ * x = (d/scale + 1)/2 ∈ [0,1]. A delta outside ±scale throws (use
863
+ * `detectScale` to pick 1 vs 100 BEFORE streaming). Default 1. */
864
+ scale?: number;
865
+ /** Seed for the data-independent shuffle of paired deltas in `decide(ctx)`
866
+ * (exchangeability guard). Default 1337. */
867
+ shuffleSeed?: number;
868
+ /** Bind the pre-registered hypothesis. Verified (content hash) at
869
+ * construction; alpha/maxN/direction/minEffect come FROM the manifest. */
870
+ preRegistration?: SignedManifest;
871
+ /** Override the gate name in reports. */
872
+ name?: string;
873
+ /** Continue the observe-stream from a `state()` snapshot taken before a
874
+ * restart. The gate's parameters still resolve from this options object
875
+ * (and the manifest); the snapshot must have been recorded under the same
876
+ * alpha, maxBet, and null boundary, and its decision must be one this
877
+ * configuration could have reached at its n, or construction throws.
878
+ * `decide(ctx)` is unaffected — it always runs on a fresh stream. */
879
+ resume?: SequentialStreamState;
880
+ }
881
+ /** Snapshot of one observe-stream: the e-process state plus the gate's own
882
+ * decision, which applies `minN` and `maxN` on top of the core latch. A
883
+ * gate rebuilt with `resume` from this value continues exactly where the
884
+ * snapshot was taken. */
885
+ type SequentialStreamState = EProcessState & {
886
+ decision: SequentialDecision;
887
+ };
888
+ interface SequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario> extends Gate<TArtifact, TScenario> {
889
+ /** Streaming entry point: feed one paired per-scenario delta
890
+ * (candidate − baseline, native scale). Each gate instance carries ONE
891
+ * observe-stream; `decide(ctx)` runs on its own fresh stream and never
892
+ * consumes or advances this one. 'promote' is sticky; observing past the
893
+ * pre-registered maxN throws (extending a finished stream after seeing
894
+ * the result reopens optional stopping — start a NEW pre-registered
895
+ * test). */
896
+ observe(delta: number): SequentialObservation;
897
+ /** Read-only snapshot of the observe-stream. Pass it back as
898
+ * `resume` to continue the stream after a restart. */
899
+ state(): SequentialStreamState;
900
+ }
901
+ /**
902
+ * Anytime-valid sequential paired gate. Conforms to the existing `Gate`
903
+ * contract (`decide(ctx)` consumes candidate vs baseline judge scores via
904
+ * `pairHoldout` — same pairing granularity as the fixed-n gates: full cellId,
905
+ * never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
906
+ * that score cells incrementally and want to stop mid-stream.
907
+ *
908
+ * Decision mapping onto the substrate's five-valued `GateDecision`:
909
+ * - 'promote' → 'ship'
910
+ * - 'continue' → 'need_more_work' (stream ended before maxN with
911
+ * the e-value undecided — more reps could decide)
912
+ * - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
913
+ * evidence of no effect (never a silent default)
914
+ */
915
+ declare function sequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: SequentialPairedGateOptions): SequentialPairedGate<TArtifact, TScenario>;
916
+ interface SequentialDecideOptions {
917
+ /** Type-I budget for the early-stop evidence. Default 0.05. */
918
+ alpha?: number;
919
+ /** Minimum paired deltas before a stop may fire. Default 5. */
920
+ minN?: number;
921
+ /** Bet truncation forwarded to `eProcess`. Default 0.5. */
922
+ maxBet?: number;
923
+ /** Bound on |per-scenario composite delta|. Default 1. */
924
+ scale?: number;
925
+ }
926
+ interface SequentialDecideFn {
927
+ (args: {
928
+ history: GenerationRecord[];
929
+ }): {
930
+ stop: boolean;
931
+ reason?: string;
932
+ };
933
+ /** Read-only snapshot of the accumulated e-process (observability + tests). */
934
+ state(): EProcessState;
935
+ }
936
+ /**
937
+ * `SurfaceProposer.decide` adapter — stops the optimization loop the moment
938
+ * the e-process decides the loop has produced a real improvement, instead of
939
+ * always running `maxGenerations`.
940
+ *
941
+ * Stream: for each generation g ≥ 1, the per-scenario composite deltas of
942
+ * generation g's top candidate vs the generation-0 top candidate (the
943
+ * incumbent the loop set out to beat), paired by scenarioId. H0: no proposed
944
+ * surface improves any scenario's expected composite over the incumbent —
945
+ * under it every delta has conditional mean ≤ 0 and the e-process is valid.
946
+ * Once wealth ≥ 1/alpha the loop stops and hands the winner to the promotion
947
+ * gate (which re-scores on HELD-OUT data — this adapter only spends the
948
+ * exploration budget, it never promotes).
949
+ *
950
+ * Honesty caveats: (1) the incumbent's scores are measured once and shared
951
+ * across all generations' deltas, so type-I control is exact only insofar as
952
+ * those scores approximate the incumbent's true per-scenario means (more reps
953
+ * → tighter); (2) an UNDECIDED process never stops the loop — absence of a
954
+ * crossing is NOT evidence of no effect, so the loop simply runs its normal
955
+ * course. Calling the adapter repeatedly with a growing history consumes each
956
+ * generation exactly once (re-feeding an already-seen record would double-count
957
+ * evidence).
958
+ */
959
+ declare function sequentialDecide(options?: SequentialDecideOptions): SequentialDecideFn;
960
+ //#endregion
961
+ //#region src/campaign/gates/statistical-heldout.d.ts
962
+ interface PairedHoldout {
963
+ /** Baseline scalar per paired cell (same order as `after`/`cellIds`). */
964
+ before: number[];
965
+ /** Candidate scalar per paired cell. */
966
+ after: number[];
967
+ /** The full cellIds (`scenario:rep`) that paired, in order. */
968
+ cellIds: string[];
969
+ }
970
+ /**
971
+ * Pair candidate vs baseline holdout observations by FULL cellId. `select`
972
+ * pulls the scalar from a cell's judge reports (composite, or a named
973
+ * dimension); a cell contributes the mean of `select` across its judges. Cells
974
+ * whose scenario is not in `scenarioIds`, or where `select` is undefined for
975
+ * every judge on either side, are skipped on BOTH sides so the arrays stay
976
+ * paired. Throws when the two maps disagree on which holdout cells exist — a
977
+ * load-bearing invariant: the baseline + winner holdout campaigns run the same
978
+ * scenarios with the same seed base, so their cellIds MUST align; a mismatch
979
+ * means a silent pairing bug, not a soft fallback.
980
+ */
981
+ declare function pairHoldout(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, select: (s: JudgeScore) => number | undefined): PairedHoldout;
982
+ interface HeldoutSignificance {
983
+ paired: PairedHoldout;
984
+ /**
985
+ * The paired bootstrap on the requested statistic (MEAN by default — see the
986
+ * tie note on `heldoutSignificance`).
987
+ *
988
+ * DIAGNOSTIC, not necessarily the interval the verdict keyed on. On a
989
+ * two-point (pass/fail) outcome the decision routes to Tango's score interval
990
+ * instead, because a percentile bootstrap of the mean over a three-atom
991
+ * lattice is not a valid interval at a nonzero margin. Read
992
+ * `decision.low`/`decision.high` for the interval that actually decided, and
993
+ * `decisionStatistic` for which one it is.
994
+ */
995
+ bootstrap: PairedBootstrapResult;
996
+ /** The MEDIAN paired-delta bootstrap, reported as a diagnostic. When many
997
+ * scenarios are tied (both sides solve them), the median is pinned near 0
998
+ * regardless of the mean lift — comparing the two exposes tie-domination. */
999
+ medianBootstrap: PairedBootstrapResult;
1000
+ /**
1001
+ * The full promotion decision: which estimator the outcome's shape admits,
1002
+ * the interval it produced, McNemar's exact veto on the two-point path, and
1003
+ * whether the interval was zero-width (no evidence in either direction). The
1004
+ * single source of `significant`.
1005
+ */
1006
+ decision: PairedPromotionDecision;
1007
+ /** Which paired estimator the verdict was decided on. */
1008
+ decisionStatistic: PairedDecisionStatistic;
1009
+ /** McNemar's exact evidence on the two-point path; null otherwise. */
1010
+ mcnemar: PairedMcNemarEvidence | null;
1011
+ /** Fraction of paired observations that are exact ties (|delta| < 1e-9). A
1012
+ * high tie fraction is WHY a median-based gate would have missed a real lift;
1013
+ * it is the observability the tie fix adds. */
1014
+ tieFraction: number;
1015
+ /** n paired observations. */
1016
+ n: number;
1017
+ /** Effective minimum after applying the bootstrap's hard statistical floor. */
1018
+ minimumRequired: number;
1019
+ /** Statistical method that carried the decision. */
1020
+ decisionMethod: PairedDecisionMethod;
1021
+ /** Exact one-sided p-value on the small-sample path; otherwise null. */
1022
+ pValue: number | null;
1023
+ /** True iff n >= minimumRequired, the DECIDING interval has nonzero width,
1024
+ * its lower bound clears the threshold, and McNemar's exact test does not
1025
+ * veto at a non-negative threshold. */
1026
+ significant: boolean;
1027
+ /** Set when n < minimumRequired — too little evidence to claim significance. */
1028
+ fewRuns: boolean;
1029
+ }
1030
+ interface HeldoutSignificanceOptions {
1031
+ deltaThreshold?: number;
1032
+ minProductiveRuns?: number;
1033
+ confidence?: number;
1034
+ resamples?: number;
1035
+ /** Fixed by default for a deterministic, reproducible gate verdict. */
1036
+ seed?: number;
1037
+ statistic?: 'mean' | 'median';
1038
+ }
1039
+ /**
1040
+ * Significance of the held-out composite lift: ship only when the lower bound
1041
+ * of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
1042
+ * 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
1043
+ * scale.
1044
+ *
1045
+ * The decision is delegated whole to {@link decidePairedPromotion}, the one
1046
+ * copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
1047
+ * also calls. That module's header carries the measurements; the short version
1048
+ * is three guards a bare `bootstrap.low > threshold` does not have:
1049
+ *
1050
+ * - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
1051
+ * only paired-binary construction that stays valid at a nonzero margin;
1052
+ * - McNemar's exact test VETOES at any non-negative threshold;
1053
+ * - a ZERO-WIDTH interval is refused rather than promoted, in either
1054
+ * direction — [0,0] clears every negative threshold and [g,g] clears every
1055
+ * threshold below g, and both are an absence of evidence, not a result.
1056
+ *
1057
+ * Measured on this function before those guards landed, at a nominal 5 %:
1058
+ * 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
1059
+ * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
1060
+ * delta is exactly 0.
1061
+ *
1062
+ * At small n, where the percentile bootstrap is descriptive only, a
1063
+ * pre-registered exact sign test still carries the bootstrap path.
1064
+ */
1065
+ declare function heldoutSignificance(paired: PairedHoldout, opts?: HeldoutSignificanceOptions): HeldoutSignificance;
1066
+ interface DimensionRegression {
1067
+ dimension: string;
1068
+ /** Paired bootstrap on (candidate − baseline). DIAGNOSTIC on a pass/fail
1069
+ * dimension, where `ci` carries the interval that decided instead. */
1070
+ bootstrap: PairedBootstrapResult;
1071
+ /** Which paired statistic `bootstrap.low` is the lower bound of. `'mean'`
1072
+ * unless the caller asked for the median. `bootstrap.median` still carries
1073
+ * the median point estimate either way. */
1074
+ bootstrapStatistic: 'median' | 'mean';
1075
+ /** The interval `regressed` was decided on, in the dimension's native units. */
1076
+ ci: {
1077
+ low: number;
1078
+ high: number;
1079
+ };
1080
+ /** Which estimator produced `ci`. */
1081
+ decisionStatistic: PairedDecisionStatistic;
1082
+ /** McNemar's exact evidence on a pass/fail dimension; null otherwise. */
1083
+ mcnemar: PairedMcNemarEvidence | null;
1084
+ /** `ci` has zero width — no evidence in either direction. */
1085
+ indeterminate: boolean;
1086
+ /** True iff the candidate may have regressed this dimension by more than
1087
+ * tolerance: the lower bound of the DECIDING interval on (candidate −
1088
+ * baseline) is below −tolerance, OR the exact small-sample test proves a drop
1089
+ * past tolerance. */
1090
+ regressed: boolean;
1091
+ tolerance: number;
1092
+ n: number;
1093
+ }
1094
+ /** Detect the native scale of a set of scores: 0-100 when any magnitude clears
1095
+ * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
1096
+ * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
1097
+ declare function detectScale(values: number[]): 1 | 100;
1098
+ /** Per-critical-dimension regression guard. For each dimension, pair the
1099
+ * candidate vs baseline values by full cellId and bootstrap the paired delta;
1100
+ * a dimension is "regressed" when the CI lower bound < −tolerance (conservative
1101
+ * — blocks if the credible worst case exceeds tolerance, which is the right
1102
+ * posture for safety dimensions like `hallucination_free`). When `tolerance`
1103
+ * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
1104
+ *
1105
+ * The interval comes from {@link decidePairedPromotion}, so a pass/fail
1106
+ * dimension is judged on Tango's score interval rather than a percentile
1107
+ * bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
1108
+ * is not a valid interval at one. That matters most here because this guard
1109
+ * fails OPEN by construction: `tolerance` is positive, so an interval pinned at
1110
+ * [0,0] never satisfies `low < −tolerance` and a real regression on a safety
1111
+ * dimension would be reported as `regressed: false`. On the median it fails the
1112
+ * same way for the same reason — when most pairs tie, which is automatic for a
1113
+ * pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
1114
+ * to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
1115
+ * restore the pre-0.134 behaviour. */
1116
+ declare function dimensionRegressions(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, criticalDimensions: string[], opts?: {
1117
+ tolerance?: number;
1118
+ confidence?: number;
1119
+ resamples?: number;
1120
+ seed?: number;
1121
+ /** Paired statistic the CI is computed on. Default `'mean'` — see
1122
+ * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */
1123
+ statistic?: 'mean' | 'median';
1124
+ }): DimensionRegression[];
1125
+ //#endregion
1126
+ export { paretoSignificanceGate as $, PairArmsOptions as A, PowerPreflight as B, hashJson as C, ComparePairedArmsOptions as D, verifyManifest as E, PairedCorrectness as F, BuildEvidenceVectorOptions as G, powerPreflight as H, PairedMetricDelta as I, ParetoSignificanceGateOptions as J, EvidenceVector as K, comparePairedArms as L, PairRunRecordsResult as M, PairedArmRow as N, MatchedPair as O, PairedArmsComparison as P, paretoPolicy as Q, pairArms as R, evaluateHypothesis as S, signManifest as T, AxisEvidence as U, PowerPreflightOptions as V, AxisVerdict as W, PromotionPolicy as X, PromotionObjective as Y, buildEvidenceVector as Z, sequentialPairedGate as _, pairedRiskDifference as _t, detectScale as a, EProcessOptions as at, SignedManifest as b, passAtK as bt, pairHoldout as c, eProcess as ct, SequentialDecision as d, ProportionInterval as dt, Objective as et, SequentialObservation as f, RiskDifferenceResult as ft, sequentialDecide as g, pairedBinaryScale as gt, SequentialStreamState as h, mcnemar as ht, PairedHoldout as i, EProcess as it, PairArmsResult as j, MatchedRunRecordPair as k, SequentialDecideFn as l, ExactRiskDifferenceResult as lt, SequentialPairedGateOptions as m, isBinaryOutcomeVector as mt, HeldoutSignificance as n, dominates as nt, dimensionRegressions as o, EProcessState as ot, SequentialPairedGate as p, ScoreRiskDifferenceResult as pt, ObjectiveSource as q, HeldoutSignificanceOptions as r, paretoFrontier as rt, heldoutSignificance as s, EProcessStep as st, DimensionRegression as t, ParetoResult as tt, SequentialDecideOptions as u, McNemarResult as ut, HypothesisManifest as v, pairedRiskDifferenceExact as vt, manifestContentDigest as w, SignedManifestAlgo as x, wilson as xt, HypothesisResult as y, pairedRiskDifferenceScore as yt, pairRunRecords as z };
1127
+ //# sourceMappingURL=statistical-heldout-Cqb73yE9.d.ts.map