@tangle-network/agent-eval 0.173.3 → 0.174.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +31 -0
- package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
- package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
- package/dist/adapters/http.d.ts +2 -2
- package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
- package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +7 -9
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
- package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
- package/dist/{benchmark-command-BY9oscke.js → benchmark-command-mZIlR-ra.js} +13 -13
- package/dist/{benchmark-command-BY9oscke.js.map → benchmark-command-mZIlR-ra.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +3 -4
- package/dist/benchmarks/index.d.ts.map +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +5 -9
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
- package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
- package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -10
- package/dist/contract/index.js +8 -8
- package/dist/{default-registry-B0bKikCb.js → default-registry-CrAp0pYq.js} +4 -4
- package/dist/{default-registry-B0bKikCb.js.map → default-registry-CrAp0pYq.js.map} +1 -1
- package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
- package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
- package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
- package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
- package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
- package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
- package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
- package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
- package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +1 -4
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
- package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/fuzz.js.map +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
- package/dist/index-BTrx5s8m.d.ts.map +1 -0
- package/dist/{index-D0Db5X-4.d.ts → index-Bn-nlnSV.d.ts} +4 -4
- package/dist/{index-D0Db5X-4.d.ts.map → index-Bn-nlnSV.d.ts.map} +1 -1
- package/dist/index-DKXuBPXf.d.ts +3840 -0
- package/dist/index-DKXuBPXf.d.ts.map +1 -0
- package/dist/index.d.ts +11 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +10 -10
- package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
- package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
- package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
- package/dist/llm-judge-DmNaBrXB.js.map +1 -0
- package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
- package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
- package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
- package/dist/rl.d.ts +1 -1
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js.map +1 -1
- package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
- package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
- package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
- package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
- package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
- package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
- package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
- package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
- package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
- package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
- package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
- package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
- package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
- package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
- package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/traces.js +4 -4
- package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
- package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
- package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
- package/dist/types-DQ0e2E7y.js.map +1 -0
- package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
- package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
- package/docs/campaign-proposers.md +42 -0
- package/package.json +1 -1
- package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
- package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
- package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
- package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
- package/dist/benchmark-BjLGkfnN.d.ts +0 -236
- package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
- package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
- package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
- package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
- package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
- package/dist/index-BQqOjerE.d.ts.map +0 -1
- package/dist/index-CFDffsKz.d.ts +0 -1135
- package/dist/index-CFDffsKz.d.ts.map +0 -1
- package/dist/llm-judge-BfqMFo4h.js.map +0 -1
- package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
- package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
- package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
- package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
- package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
- package/dist/proposal-findings-bko3GGy-.js.map +0 -1
- package/dist/provenance-CRY67X50.d.ts +0 -1995
- package/dist/provenance-CRY67X50.d.ts.map +0 -1
- package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
- package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
- package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
- package/dist/types-CiWITkGo.js.map +0 -1
|
@@ -1,592 +0,0 @@
|
|
|
1
|
-
import { a as RunRecord } from "./run-record-DTv1MdjK.js";
|
|
2
|
-
import { d as PairedBootstrapResult, u as PairedBootstrapOptions } from "./paired-promotion-decision-CGzg0cI_.js";
|
|
3
|
-
//#region src/statistics/paired-binary.d.ts
|
|
4
|
-
/** A binomial proportion estimate with a confidence interval. */
|
|
5
|
-
interface ProportionInterval {
|
|
6
|
-
/** Point estimate successes / n (0 when n = 0). */
|
|
7
|
-
estimate: number;
|
|
8
|
-
/** Lower bound, clamped to [0, 1]. */
|
|
9
|
-
lower: number;
|
|
10
|
-
/** Upper bound, clamped to [0, 1]. */
|
|
11
|
-
upper: number;
|
|
12
|
-
}
|
|
13
|
-
/**
|
|
14
|
-
* Wilson score interval for a binomial proportion. Correct at small n and near
|
|
15
|
-
* 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
|
|
16
|
-
* understates coverage. Use this for any pass-rate / hit-rate / realness-rate
|
|
17
|
-
* CI — the continuous `confidenceInterval` assumes the wrong distribution for a
|
|
18
|
-
* proportion. `n = 0 ⇒ {0, 0, 0}`.
|
|
19
|
-
*/
|
|
20
|
-
declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
|
|
21
|
-
/**
|
|
22
|
-
* Are these per-item outcomes binary (every value exactly 0 or 1)?
|
|
23
|
-
*
|
|
24
|
-
* The discriminator a promotion gate needs before choosing a paired statistic.
|
|
25
|
-
* On binary outcomes the paired delta vector lives in {-1, 0, +1} and is
|
|
26
|
-
* normally dominated by zeros (both arms solve, or both arms miss, most items),
|
|
27
|
-
* so its MEDIAN is pinned at exactly 0 no matter how large the real shift in
|
|
28
|
-
* success rate is — and a bootstrap CI on that median collapses to [0, 0].
|
|
29
|
-
* A gate keying on `ci.low > threshold` is then structurally unable to see
|
|
30
|
-
* either a gain or a regression. Detect this shape and switch to the
|
|
31
|
-
* paired-binary estimators ({@link mcnemar}, {@link pairedRiskDifference})
|
|
32
|
-
* instead of silently answering "no" forever.
|
|
33
|
-
*
|
|
34
|
-
* Empty input is NOT binary: there is no evidence of the outcome's shape, and
|
|
35
|
-
* defaulting an empty vector into the binary branch would pick a statistic on
|
|
36
|
-
* no data at all.
|
|
37
|
-
*
|
|
38
|
-
* NOT the right discriminator for a gate. It recognises the literal {0, 1}
|
|
39
|
-
* encoding and nothing else, so a pass/fail dimension emitted on 0-100 — which
|
|
40
|
-
* judges in this codebase do routinely — reads as non-binary, and a single
|
|
41
|
-
* partial-credit score in an otherwise pass/fail vector flips it to false while
|
|
42
|
-
* leaving the median just as blind. Gates want {@link pairedBinaryScale} (any
|
|
43
|
-
* two-point encoding). This predicate remains for callers that specifically
|
|
44
|
-
* mean "literally 0/1".
|
|
45
|
-
*/
|
|
46
|
-
declare function isBinaryOutcomeVector(values: ArrayLike<number>): boolean;
|
|
47
|
-
/** Result of a McNemar paired-binary significance test. */
|
|
48
|
-
interface McNemarResult {
|
|
49
|
-
/** Total paired observations. */
|
|
50
|
-
n: number;
|
|
51
|
-
/** Discordant pairs (b + c) — the only ones that carry signal. */
|
|
52
|
-
nDiscordant: number;
|
|
53
|
-
/** Pairs where treatment succeeded and control failed ("newly correct"). */
|
|
54
|
-
b: number;
|
|
55
|
-
/** Pairs where control succeeded and treatment failed ("newly wrong"). */
|
|
56
|
-
c: number;
|
|
57
|
-
/** Continuity-corrected chi-square statistic (reference; exact p drives the call). */
|
|
58
|
-
statistic: number;
|
|
59
|
-
/** Two-sided p-value. Exact (binomial sign test on discordant pairs). */
|
|
60
|
-
pValue: number;
|
|
61
|
-
}
|
|
62
|
-
/**
|
|
63
|
-
* McNemar's test for paired binary outcomes — the correct significance test for
|
|
64
|
-
* "does treatment change the success rate vs control on the SAME items". Only
|
|
65
|
-
* discordant pairs (one arm right, the other wrong) carry information; concordant
|
|
66
|
-
* pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
|
|
67
|
-
* rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
|
|
68
|
-
* Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
|
|
69
|
-
* at the small discordant counts typical of eval runs (no continuity-corrected
|
|
70
|
-
* chi-square approximation needed, though it is returned as `statistic` for
|
|
71
|
-
* reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
|
|
72
|
-
* the module's (before, after) convention. Throws on unequal lengths.
|
|
73
|
-
*/
|
|
74
|
-
declare function mcnemar(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>): McNemarResult;
|
|
75
|
-
/** A paired binary effect size (treatment rate − control rate) with a CI. */
|
|
76
|
-
interface RiskDifferenceResult {
|
|
77
|
-
/** Total paired observations. */
|
|
78
|
-
n: number;
|
|
79
|
-
/** Discordant pairs: treatment-win count. */
|
|
80
|
-
b: number;
|
|
81
|
-
/** Discordant pairs: control-win count. */
|
|
82
|
-
c: number;
|
|
83
|
-
/** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
|
|
84
|
-
riskDifference: number;
|
|
85
|
-
/** Lower bound of the CI, clamped to [-1, 1]. */
|
|
86
|
-
lower: number;
|
|
87
|
-
/** Upper bound of the CI, clamped to [-1, 1]. */
|
|
88
|
-
upper: number;
|
|
89
|
-
/** Confidence level used. */
|
|
90
|
-
confidence: number;
|
|
91
|
-
}
|
|
92
|
-
/**
|
|
93
|
-
* Paired risk difference (the effect-size companion to {@link mcnemar}): the
|
|
94
|
-
* change in success rate p(treatment) − p(control) on matched items, which for
|
|
95
|
-
* paired binary data equals (b − c) / n. The CI uses the paired variance from
|
|
96
|
-
* the discordant counts, not the independent-samples formula (which overstates
|
|
97
|
-
* the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
|
|
98
|
-
* arrays, control first. Throws on unequal lengths.
|
|
99
|
-
*
|
|
100
|
-
* REPORTING ONLY — do NOT decide a promotion on this interval. The CI is a Wald
|
|
101
|
-
* normal approximation, which badly UNDERCOVERS when only a handful of pairs are
|
|
102
|
-
* discordant: at n = 3 with b = 2, c = 0 it returns [0.133, 1.000], excluding 0,
|
|
103
|
-
* while McNemar's exact test on the same data gives p = 0.50. A gate keying on
|
|
104
|
-
* `lower > 0` would promote noise. Use {@link pairedRiskDifferenceExact}, whose
|
|
105
|
-
* interval is dual to the exact test by construction, for any decision.
|
|
106
|
-
*/
|
|
107
|
-
declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
|
|
108
|
-
/** A paired binary effect size with an EXACT interval and the exact test that
|
|
109
|
-
* bounds it — one object so a caller cannot read the estimate without the
|
|
110
|
-
* significance it is entitled to. */
|
|
111
|
-
interface ExactRiskDifferenceResult {
|
|
112
|
-
/** Total paired observations. */
|
|
113
|
-
n: number;
|
|
114
|
-
/** Discordant pairs: treatment-win count. */
|
|
115
|
-
b: number;
|
|
116
|
-
/** Discordant pairs: control-win count. */
|
|
117
|
-
c: number;
|
|
118
|
-
/** Discordant pairs (b + c) — the only ones carrying information. */
|
|
119
|
-
nDiscordant: number;
|
|
120
|
-
/** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
|
|
121
|
-
riskDifference: number;
|
|
122
|
-
/** Exact conditional CI lower bound. 0 when there are no discordant pairs. */
|
|
123
|
-
lower: number;
|
|
124
|
-
/** Exact conditional CI upper bound. 0 when there are no discordant pairs. */
|
|
125
|
-
upper: number;
|
|
126
|
-
/** Confidence level used. */
|
|
127
|
-
confidence: number;
|
|
128
|
-
/** McNemar's exact two-sided p-value on the same discordant counts. */
|
|
129
|
-
pValue: number;
|
|
130
|
-
}
|
|
131
|
-
/**
|
|
132
|
-
* Paired risk difference with the EXACT CONDITIONAL interval — the estimator a
|
|
133
|
-
* promotion gate may decide on.
|
|
134
|
-
*
|
|
135
|
-
* Conditional on the number of discordant pairs m = b + c, the treatment-win
|
|
136
|
-
* count b is Binomial(m, π) with π = P(treatment wins | discordant), and the
|
|
137
|
-
* risk difference is an exact reparameterisation: RD = (2π − 1)·m/n. So a
|
|
138
|
-
* Clopper-Pearson exact interval for π maps straight onto RD. This buys the
|
|
139
|
-
* property the Wald interval in {@link pairedRiskDifference} does not have:
|
|
140
|
-
*
|
|
141
|
-
* **`lower > 0` ⟺ McNemar's exact test rejects at α = 1 − confidence.**
|
|
142
|
-
*
|
|
143
|
-
* Clopper-Pearson excludes π = 0.5 exactly when the two-sided exact binomial
|
|
144
|
-
* test of π = 0.5 rejects, and that test IS {@link mcnemar}'s p-value — so the
|
|
145
|
-
* interval and the test can never disagree, and a gate keyed on `lower` cannot
|
|
146
|
-
* promote what the exact test refuses. The exact p is returned in the same
|
|
147
|
-
* object so the two are impossible to compute apart.
|
|
148
|
-
*
|
|
149
|
-
* The interval is conservative (exact intervals over-cover; conditioning on m
|
|
150
|
-
* discards the concordant pairs' information about m itself). That is the
|
|
151
|
-
* correct direction for a promotion gate: it refuses more often, never less.
|
|
152
|
-
*
|
|
153
|
-
* With m = 0 there are no discordant pairs and π is not identified: the result
|
|
154
|
-
* is the degenerate [0, 0] with p = 1. That is NOT evidence of equivalence —
|
|
155
|
-
* callers must treat a zero-width interval as "cannot decide", not as "no
|
|
156
|
-
* difference". Inputs are paired 0/1 (or boolean) arrays, control first.
|
|
157
|
-
* Throws on unequal lengths.
|
|
158
|
-
*/
|
|
159
|
-
declare function pairedRiskDifferenceExact(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ExactRiskDifferenceResult;
|
|
160
|
-
/** A paired binary effect size with an interval that is valid at a NONZERO
|
|
161
|
-
* margin — the estimator a noninferiority decision may be made on. */
|
|
162
|
-
interface ScoreRiskDifferenceResult {
|
|
163
|
-
/** Total paired observations. */
|
|
164
|
-
n: number;
|
|
165
|
-
/** Discordant pairs: treatment-win count. */
|
|
166
|
-
b: number;
|
|
167
|
-
/** Discordant pairs: control-win count. */
|
|
168
|
-
c: number;
|
|
169
|
-
/** Discordant pairs (b + c). */
|
|
170
|
-
nDiscordant: number;
|
|
171
|
-
/** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
|
|
172
|
-
riskDifference: number;
|
|
173
|
-
/** Score-interval lower bound on the population risk difference. */
|
|
174
|
-
lower: number;
|
|
175
|
-
/** Score-interval upper bound on the population risk difference. */
|
|
176
|
-
upper: number;
|
|
177
|
-
/** Confidence level used. */
|
|
178
|
-
confidence: number;
|
|
179
|
-
}
|
|
180
|
-
/**
|
|
181
|
-
* Paired risk difference with TANGO'S (1998) SCORE INTERVAL — the estimator a
|
|
182
|
-
* promotion gate may decide on **at a nonzero margin**.
|
|
183
|
-
*
|
|
184
|
-
* {@link pairedRiskDifferenceExact} conditions on the observed discordant count
|
|
185
|
-
* `m = b + c`, builds a Clopper-Pearson interval for the win share among those
|
|
186
|
-
* `m` pairs, and multiplies by the observed `m/n`. That is exact for testing
|
|
187
|
-
* RD = 0 — it is dual to McNemar — but it is NOT a confidence interval for the
|
|
188
|
-
* population risk difference at a nonzero margin, because the sampling
|
|
189
|
-
* variability of `m/n` itself is discarded. The gap is not academic: with the
|
|
190
|
-
* production caller's `pairedDeltaThreshold: -0.05`, a process whose true risk
|
|
191
|
-
* difference sits exactly on that margin clears a nominal-95 % `lower > margin`
|
|
192
|
-
* check 24.75 % of the time at n = 40 and 43.95 % at n = 76 (2000 replicates
|
|
193
|
-
* each) when the conditional interval decides.
|
|
194
|
-
*
|
|
195
|
-
* Tango's interval inverts the score test of RD = delta, which estimates the
|
|
196
|
-
* nuisance loss rate under each hypothesised delta instead of fixing it at the
|
|
197
|
-
* observed value, so `m` contributes its own uncertainty. It is the method
|
|
198
|
-
* `ratesci::scorepairci` uses for paired risk-difference noninferiority, and it
|
|
199
|
-
* is not conditional, so it stays valid as the margin moves away from zero.
|
|
200
|
-
*
|
|
201
|
-
* The bounds are found by bisecting `tangoScore(delta) = ±z` — the score is
|
|
202
|
-
* monotone decreasing in delta, so each crossing is unique. Inputs are paired
|
|
203
|
-
* 0/1 (or boolean) arrays, control first. Throws on unequal lengths.
|
|
204
|
-
*/
|
|
205
|
-
declare function pairedRiskDifferenceScore(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ScoreRiskDifferenceResult;
|
|
206
|
-
/**
|
|
207
|
-
* The common positive level `s` such that EVERY value across both paired arms is
|
|
208
|
-
* exactly 0 or `s` — i.e. the outcome is pass/fail, whatever encoding it arrived
|
|
209
|
-
* in. Returns null when the outcomes are not two-point, when the two arms use
|
|
210
|
-
* different levels, or when no positive value was observed at all (all-zero
|
|
211
|
-
* arms: the level is not identified, and there is nothing to decide anyway).
|
|
212
|
-
*
|
|
213
|
-
* This is the scale-aware successor to {@link isBinaryOutcomeVector}, which only
|
|
214
|
-
* recognises literal {0, 1}. Judges in this codebase emit dimensions on 0-100 as
|
|
215
|
-
* well as [0,1] (see `detectScale` in `campaign/gates/statistical-heldout.ts`),
|
|
216
|
-
* so a pass/fail dimension routinely arrives as {0, 100} and a {0,1}-only test
|
|
217
|
-
* silently sends it down the median path that cannot see it. Any positive level
|
|
218
|
-
* is accepted, not just 1 and 100: for a two-point {0, s} outcome the mean paired
|
|
219
|
-
* delta is exactly s·(b − c)/n, so the binary estimators apply after dividing by
|
|
220
|
-
* s and rescaling the result back into the caller's native units.
|
|
221
|
-
*
|
|
222
|
-
* Non-finite values ⇒ null: an unusable outcome must not be classified as a
|
|
223
|
-
* clean pass/fail shape.
|
|
224
|
-
*/
|
|
225
|
-
declare function pairedBinaryScale(before: ArrayLike<number>, after: ArrayLike<number>): number | null;
|
|
226
|
-
/**
|
|
227
|
-
* Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
|
|
228
|
-
* Language Models Trained on Code"). Given `n` independent samples for one
|
|
229
|
-
* problem of which `c` pass, the probability that at least one of a random k of
|
|
230
|
-
* them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
|
|
231
|
-
* first k pass" is biased high at small n; this is the variance-reduced estimator
|
|
232
|
-
* averaged implicitly over all k-subsets. Average the per-problem values across
|
|
233
|
-
* the suite for the corpus pass@k. Computed in the numerically stable product
|
|
234
|
-
* form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
|
|
235
|
-
*/
|
|
236
|
-
declare function passAtK(n: number, c: number, k: number): number;
|
|
237
|
-
//#endregion
|
|
238
|
-
//#region src/statistics/sequential-eprocess.d.ts
|
|
239
|
-
interface EProcessOptions {
|
|
240
|
-
/** Type-I error budget. The process decides when wealth ≥ 1/alpha
|
|
241
|
-
* (Ville's inequality). Default 0.05. */
|
|
242
|
-
alpha?: number;
|
|
243
|
-
/** Truncation bound on the predictable bet λ ∈ [0, maxBet]. Must satisfy
|
|
244
|
-
* maxBet < 1/nullMean so every wealth factor stays strictly positive.
|
|
245
|
-
* Default 0.5. */
|
|
246
|
-
maxBet?: number;
|
|
247
|
-
/** The null boundary m₀ for H0: E[x] ≤ m₀ on x ∈ [0,1]. Default 0.5
|
|
248
|
-
* (the paired-delta encoding x = (d+1)/2 maps "no effect" to 1/2).
|
|
249
|
-
* A pre-registered minEffect shifts this — see `sequentialPairedGate`. */
|
|
250
|
-
nullMean?: number;
|
|
251
|
-
/** Continue a process from a `state()` snapshot taken before a restart.
|
|
252
|
-
* The snapshot never supplies the parameters: `alpha`, `maxBet`, and
|
|
253
|
-
* `nullMean` resolve from this options object exactly as for a fresh
|
|
254
|
-
* process, and a snapshot recorded under different parameters is refused
|
|
255
|
-
* (`ValidationError`) — re-deciding the same stream under new parameters
|
|
256
|
-
* would reopen optional stopping. A snapshot whose fields are not
|
|
257
|
-
* mutually consistent (tampered or truncated) is refused the same way. */
|
|
258
|
-
resume?: EProcessState;
|
|
259
|
-
}
|
|
260
|
-
interface EProcessStep {
|
|
261
|
-
/** Current wealth W_n — the e-value against H0 after n observations. */
|
|
262
|
-
wealth: number;
|
|
263
|
-
/** Observations consumed so far. */
|
|
264
|
-
n: number;
|
|
265
|
-
/** True from the first n where W_n ≥ 1/alpha onward (sticky). */
|
|
266
|
-
decided: boolean;
|
|
267
|
-
}
|
|
268
|
-
/**
|
|
269
|
-
* Complete snapshot of an e-process. Together with the parameters it is
|
|
270
|
-
* sufficient to continue the process after a restart: `sumX` and `varSum`
|
|
271
|
-
* are the running sums the next bet is computed from, so a process rebuilt
|
|
272
|
-
* from a snapshot produces the same wealth sequence and decision as one that
|
|
273
|
-
* was never interrupted. Plain data: a JSON round-trip preserves it
|
|
274
|
-
* (`decidedAtN` is undefined, and therefore omitted, until decided).
|
|
275
|
-
*/
|
|
276
|
-
interface EProcessState extends EProcessStep {
|
|
277
|
-
alpha: number;
|
|
278
|
-
maxBet: number;
|
|
279
|
-
nullMean: number;
|
|
280
|
-
/** The decision boundary 1/alpha. */
|
|
281
|
-
threshold: number;
|
|
282
|
-
/** Observation count at the first threshold crossing; undefined until decided. */
|
|
283
|
-
decidedAtN?: number;
|
|
284
|
-
/** Σ x_i over the n observations consumed. */
|
|
285
|
-
sumX: number;
|
|
286
|
-
/** Σ (x_i − μ̂_i)² over the n observations consumed, μ̂_i the shrunk running
|
|
287
|
-
* mean after observation i. */
|
|
288
|
-
varSum: number;
|
|
289
|
-
}
|
|
290
|
-
interface EProcess {
|
|
291
|
-
/** Consume one observation x ∈ [0,1]. Throws on non-finite / out-of-range
|
|
292
|
-
* input — a silent clamp would corrupt the type-I guarantee. */
|
|
293
|
-
update(x: number): EProcessStep;
|
|
294
|
-
state(): EProcessState;
|
|
295
|
-
}
|
|
296
|
-
/**
|
|
297
|
-
* Betting test-martingale for bounded observations — the e-process core of
|
|
298
|
-
* anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
|
|
299
|
-
* of bounded random variables by betting", JRSS-B 2024).
|
|
300
|
-
*
|
|
301
|
-
* Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
|
|
302
|
-
*
|
|
303
|
-
* W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
|
|
304
|
-
*
|
|
305
|
-
* with the truncated GROW-style plug-in bet computed from PRIOR observations:
|
|
306
|
-
*
|
|
307
|
-
* λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
|
|
308
|
-
*
|
|
309
|
-
* where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
|
|
310
|
-
* σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
|
|
311
|
-
*
|
|
312
|
-
* PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
|
|
313
|
-
* ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
|
|
314
|
-
* E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
|
|
315
|
-
* supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
|
|
316
|
-
* type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
|
|
317
|
-
* (no prior evidence), so the first observation never moves wealth.
|
|
318
|
-
*
|
|
319
|
-
* `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
|
|
320
|
-
* wealth keeps updating after the crossing (the e-process remains valid), but
|
|
321
|
-
* the decision time is the first crossing.
|
|
322
|
-
*/
|
|
323
|
-
declare function eProcess(opts?: EProcessOptions): EProcess;
|
|
324
|
-
//#endregion
|
|
325
|
-
//#region src/paired-arms.d.ts
|
|
326
|
-
/** One arm observation of one work item. Structural on purpose: callers
|
|
327
|
-
* project their own record type (e.g. a `RunRecord`) into this shape. */
|
|
328
|
-
interface PairedArmRow {
|
|
329
|
-
/** Matching key — rows sharing a `pairKey` across both arms form pairs
|
|
330
|
-
* (typically the task/scenario/seed identity). */
|
|
331
|
-
pairKey: string;
|
|
332
|
-
/** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on
|
|
333
|
-
* every row of a `pairKey` that has more than one rep in either arm; reps
|
|
334
|
-
* then pair only on exact (`pairKey`, `repKey`) match, never on outcome
|
|
335
|
-
* content. Optional when each arm has at most one rep of the item. */
|
|
336
|
-
repKey?: string;
|
|
337
|
-
/** Arm label this row was produced under. */
|
|
338
|
-
arm: string;
|
|
339
|
-
/** Binary outcome; omit when the comparison has no pass/fail notion. */
|
|
340
|
-
pass?: boolean;
|
|
341
|
-
/** Named numeric measurements (score, cost, latency, …). */
|
|
342
|
-
metrics?: Record<string, number>;
|
|
343
|
-
}
|
|
344
|
-
interface PairArmsOptions {
|
|
345
|
-
/** Arm treated as the control side of every pair. */
|
|
346
|
-
baselineArm: string;
|
|
347
|
-
/** Arm treated as the treatment side of every pair. */
|
|
348
|
-
treatmentArm: string;
|
|
349
|
-
}
|
|
350
|
-
/** One matched (baseline, treatment) observation of the same work item. */
|
|
351
|
-
interface MatchedPair {
|
|
352
|
-
pairKey: string;
|
|
353
|
-
/** 0-based position of this pair within its `pairKey`, ordered by sorted
|
|
354
|
-
* `repKey` (always 0 for a single-rep item). The rep identity itself is on
|
|
355
|
-
* the rows (`baseline.repKey` / `treatment.repKey`). */
|
|
356
|
-
repIndex: number;
|
|
357
|
-
baseline: PairedArmRow;
|
|
358
|
-
treatment: PairedArmRow;
|
|
359
|
-
}
|
|
360
|
-
interface PairArmsResult {
|
|
361
|
-
/** Matched pairs, ordered by (`pairKey`, `repIndex`). */
|
|
362
|
-
pairs: MatchedPair[];
|
|
363
|
-
/** Baseline rows left without a treatment counterpart — reported, never
|
|
364
|
-
* silently dropped. */
|
|
365
|
-
unpairedBaseline: PairedArmRow[];
|
|
366
|
-
/** Treatment rows left without a baseline counterpart. */
|
|
367
|
-
unpairedTreatment: PairedArmRow[];
|
|
368
|
-
}
|
|
369
|
-
/**
|
|
370
|
-
* Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
|
|
371
|
-
*
|
|
372
|
-
* A `pairKey` with at most one row per arm pairs directly, no `repKey`
|
|
373
|
-
* needed. A `pairKey` with multiple reps in either arm requires `repKey` on
|
|
374
|
-
* every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
|
|
375
|
-
* match — pairing is keyed purely on row identity, never on outcome content
|
|
376
|
-
* (outcome-keyed matching deflates discordant counts and biases McNemar), and
|
|
377
|
-
* is therefore independent of input order. Reps whose `repKey` has no
|
|
378
|
-
* counterpart in the other arm, and items present in only one arm, land in
|
|
379
|
-
* the unpaired lists — reported, never truncated.
|
|
380
|
-
*
|
|
381
|
-
* Fail-loud: throws when either named arm has zero rows (an unknown arm
|
|
382
|
-
* name would otherwise read as "everything unpaired"), when the two arm
|
|
383
|
-
* names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
|
|
384
|
-
* when a (`pairKey`, arm) group repeats a `repKey` (the match would be
|
|
385
|
-
* ambiguous).
|
|
386
|
-
*/
|
|
387
|
-
declare function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult;
|
|
388
|
-
/** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */
|
|
389
|
-
interface PairedCorrectness {
|
|
390
|
-
/** Discordant pairs where the treatment passed and the baseline failed. */
|
|
391
|
-
b10: number;
|
|
392
|
-
/** Discordant pairs where the baseline passed and the treatment failed. */
|
|
393
|
-
b01: number;
|
|
394
|
-
/** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */
|
|
395
|
-
mcnemar: McNemarResult;
|
|
396
|
-
/** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */
|
|
397
|
-
riskDifference: RiskDifferenceResult;
|
|
398
|
-
}
|
|
399
|
-
/** Paired delta summary for one named metric (delta = treatment − baseline). */
|
|
400
|
-
interface PairedMetricDelta {
|
|
401
|
-
name: string;
|
|
402
|
-
/** Pairs where BOTH sides carry a finite value for this metric. */
|
|
403
|
-
n: number;
|
|
404
|
-
/** Pairs where at least one side does not carry the metric. */
|
|
405
|
-
nMissing: number;
|
|
406
|
-
/** Median paired delta, or null when `n === 0`. */
|
|
407
|
-
medianDelta: number | null;
|
|
408
|
-
/** Mean paired delta, or null when `n === 0`. */
|
|
409
|
-
meanDelta: number | null;
|
|
410
|
-
/** Bootstrap CI on the paired delta (`pairedBootstrap`); null when
|
|
411
|
-
* `n === 0` — a zero-width [0, 0] interval on no data would read as a
|
|
412
|
-
* measured tight null. */
|
|
413
|
-
bootstrapCi: PairedBootstrapResult | null;
|
|
414
|
-
/** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */
|
|
415
|
-
wilcoxon: {
|
|
416
|
-
w: number;
|
|
417
|
-
p: number;
|
|
418
|
-
} | null;
|
|
419
|
-
}
|
|
420
|
-
interface ComparePairedArmsOptions extends PairArmsOptions {
|
|
421
|
-
/** Metrics to compare. Default: every metric name observed on any matched
|
|
422
|
-
* pair, sorted. A name that appears on no pair is still reported (with
|
|
423
|
-
* `n = 0`) so a misspelled metric is visible instead of vanishing. */
|
|
424
|
-
metricNames?: string[];
|
|
425
|
-
/** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */
|
|
426
|
-
bootstrap?: PairedBootstrapOptions;
|
|
427
|
-
}
|
|
428
|
-
interface PairedArmsComparison {
|
|
429
|
-
nPairs: number;
|
|
430
|
-
nUnpairedBaseline: number;
|
|
431
|
-
nUnpairedTreatment: number;
|
|
432
|
-
/** null when no matched pair carries `pass` on both sides — a pass/fail
|
|
433
|
-
* verdict over rows that never measured pass/fail would be fabricated. */
|
|
434
|
-
correctness: PairedCorrectness | null;
|
|
435
|
-
metricDeltas: PairedMetricDelta[];
|
|
436
|
-
}
|
|
437
|
-
/**
|
|
438
|
-
* Full matched-pair arm comparison: pair via {@link pairArms}, then compose
|
|
439
|
-
* the paired estimators from `statistics` over the matched pairs.
|
|
440
|
-
*
|
|
441
|
-
* Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
|
|
442
|
-
* is that subset's size); each metric uses only the pairs where both sides
|
|
443
|
-
* carry a finite value for it, with the remainder counted in `nMissing`.
|
|
444
|
-
* Deltas are treatment − baseline throughout.
|
|
445
|
-
*
|
|
446
|
-
* Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
|
|
447
|
-
* non-finite metric value — silently treating corrupt telemetry as "metric
|
|
448
|
-
* absent" would misreport it as missing coverage.
|
|
449
|
-
*/
|
|
450
|
-
declare function comparePairedArms(rows: readonly PairedArmRow[], opts: ComparePairedArmsOptions): PairedArmsComparison;
|
|
451
|
-
interface MatchedRunRecordPair {
|
|
452
|
-
pairKey: string;
|
|
453
|
-
repKey: string;
|
|
454
|
-
baseline: RunRecord;
|
|
455
|
-
treatment: RunRecord;
|
|
456
|
-
}
|
|
457
|
-
interface PairRunRecordsResult {
|
|
458
|
-
pairs: MatchedRunRecordPair[];
|
|
459
|
-
unpairedBaseline: RunRecord[];
|
|
460
|
-
unpairedTreatment: RunRecord[];
|
|
461
|
-
}
|
|
462
|
-
/**
|
|
463
|
-
* Pair two RunRecord arms by the identity of the evaluated work:
|
|
464
|
-
* `(experimentId, scenarioId, seed)`.
|
|
465
|
-
*
|
|
466
|
-
* Falling back to array order, candidate id, or experiment id can compare
|
|
467
|
-
* different tasks and fabricate lift. Duplicate identities throw.
|
|
468
|
-
*/
|
|
469
|
-
declare function pairRunRecords(baselineRuns: readonly RunRecord[], treatmentRuns: readonly RunRecord[]): PairRunRecordsResult;
|
|
470
|
-
//#endregion
|
|
471
|
-
//#region src/pre-registration.d.ts
|
|
472
|
-
/**
|
|
473
|
-
* Pre-registered hypotheses — declare what you're testing BEFORE the
|
|
474
|
-
* run, check it AFTER. Prevents p-hacking, optional stopping, and the
|
|
475
|
-
* "we ran until it looked good" failure mode.
|
|
476
|
-
*
|
|
477
|
-
* Manifest is a plain JSON-friendly object. Sign it with a content hash
|
|
478
|
-
* + timestamp; the registered record becomes immutable. Post-run,
|
|
479
|
-
* evaluate the manifest against observed results — the library refuses
|
|
480
|
-
* to let you re-interpret a different metric as the declared one.
|
|
481
|
-
*
|
|
482
|
-
* A signed manifest is a portable record: it is written once and verified
|
|
483
|
-
* later, possibly by a different release. `algo` names the digest scheme it
|
|
484
|
-
* was signed under, and verification selects the encoder by that field, so a
|
|
485
|
-
* manifest signed by an earlier release still verifies.
|
|
486
|
-
*/
|
|
487
|
-
interface HypothesisManifest {
|
|
488
|
-
id: string;
|
|
489
|
-
/** Human prose — goes into the audit trail. */
|
|
490
|
-
hypothesis: string;
|
|
491
|
-
/** Metric the hypothesis claims to move. */
|
|
492
|
-
metric: string;
|
|
493
|
-
/** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
|
|
494
|
-
direction: 'increase' | 'decrease';
|
|
495
|
-
/** Minimum effect size to count (same units as the metric). */
|
|
496
|
-
minEffect: number;
|
|
497
|
-
/** Alpha threshold. */
|
|
498
|
-
alpha: number;
|
|
499
|
-
/** Target statistical power at which sample size was pre-computed. */
|
|
500
|
-
power: number;
|
|
501
|
-
/** Declared N per arm before running. */
|
|
502
|
-
preRegisteredN: number;
|
|
503
|
-
/** ISO8601 timestamp the manifest was registered. */
|
|
504
|
-
registeredAt: string;
|
|
505
|
-
/** Optional identifiers to tie into the trace corpus. */
|
|
506
|
-
baselineLabel?: string;
|
|
507
|
-
candidateLabel?: string;
|
|
508
|
-
}
|
|
509
|
-
/**
|
|
510
|
-
* Identifier for the hashing scheme used to produce `contentHash`.
|
|
511
|
-
*
|
|
512
|
-
* Both schemes are sha256 hex over the manifest with `contentHash` and `algo`
|
|
513
|
-
* stripped, and differ only in how that manifest is serialized:
|
|
514
|
-
*
|
|
515
|
-
* - `'sha256-rfc8785'` — RFC 8785 canonical JSON. What {@link signManifest}
|
|
516
|
-
* emits.
|
|
517
|
-
* - `'sha256-content'` — key-sorted `JSON.stringify`. Read-only: manifests
|
|
518
|
-
* signed by an earlier release carry it, or carry no `algo` at all, and
|
|
519
|
-
* {@link verifyManifest} still verifies them.
|
|
520
|
-
*/
|
|
521
|
-
type SignedManifestAlgo = 'sha256-content' | 'sha256-rfc8785';
|
|
522
|
-
interface SignedManifest extends HypothesisManifest {
|
|
523
|
-
/** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
|
|
524
|
-
contentHash: string;
|
|
525
|
-
/**
|
|
526
|
-
* Algorithm string describing how `contentHash` was produced.
|
|
527
|
-
*
|
|
528
|
-
* Optional on the type so serialized manifests without it still parse,
|
|
529
|
-
* but ALWAYS populated by {@link signManifest}. Consumers that want to
|
|
530
|
-
* enforce a known algorithm should reject manifests where this field
|
|
531
|
-
* is missing or unrecognized.
|
|
532
|
-
*/
|
|
533
|
-
algo?: SignedManifestAlgo;
|
|
534
|
-
}
|
|
535
|
-
interface HypothesisResult {
|
|
536
|
-
manifest: SignedManifest;
|
|
537
|
-
observedN: number;
|
|
538
|
-
observedEffect: number;
|
|
539
|
-
observedPValue: number;
|
|
540
|
-
/** True iff the observed effect hits the pre-declared direction with
|
|
541
|
-
* magnitude ≥ minEffect AND p < alpha. */
|
|
542
|
-
confirmed: boolean;
|
|
543
|
-
/** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
|
|
544
|
-
rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
|
|
545
|
-
notes?: string;
|
|
546
|
-
}
|
|
547
|
-
/**
|
|
548
|
-
* SHA-256 hex (full 64 chars) over the RFC 8785 canonical JSON encoding of
|
|
549
|
-
* `obj` — the package's one identity scheme, shared with `ledger-core`.
|
|
550
|
-
*
|
|
551
|
-
* Values canonical JSON cannot represent faithfully — `undefined`, `NaN`,
|
|
552
|
-
* class instances, cycles — are refused rather than coerced, because a
|
|
553
|
-
* coercion maps two distinct records onto one digest.
|
|
554
|
-
*
|
|
555
|
-
* Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
|
|
556
|
-
* which takes a string input and returns a truncated 12-char prompt id.
|
|
557
|
-
*
|
|
558
|
-
* @example
|
|
559
|
-
* const hash = await hashJson({ id: '1', kind: 'spec' })
|
|
560
|
-
* // 'a3f1...' (64 hex chars)
|
|
561
|
-
*/
|
|
562
|
-
declare function hashJson<T>(obj: T): Promise<string>;
|
|
563
|
-
/**
|
|
564
|
-
* Digest of a manifest under its own declared scheme, with `contentHash` and
|
|
565
|
-
* `algo` stripped. Synchronous, so a caller that must fail before consuming an
|
|
566
|
-
* observation does not have to await. Throws on an `algo` this release does
|
|
567
|
-
* not know — an unverifiable manifest must not read as a valid one.
|
|
568
|
-
*/
|
|
569
|
-
declare function manifestContentDigest(manifest: SignedManifest): string;
|
|
570
|
-
/**
|
|
571
|
-
* Sign a manifest with a SHA-256 content hash over its RFC 8785 canonical
|
|
572
|
-
* JSON, with `contentHash` and `algo` stripped, and stamp the scheme in
|
|
573
|
-
* `algo` so a later reader knows which encoder to verify with.
|
|
574
|
-
*/
|
|
575
|
-
declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
|
|
576
|
-
/**
|
|
577
|
-
* Verify that a signed manifest has not been tampered with, under the scheme
|
|
578
|
-
* the manifest itself declares.
|
|
579
|
-
*/
|
|
580
|
-
declare function verifyManifest(m: SignedManifest): Promise<boolean>;
|
|
581
|
-
/**
|
|
582
|
-
* Evaluate a pre-registered hypothesis against observed results.
|
|
583
|
-
* Mechanical — no re-interpretation permitted.
|
|
584
|
-
*/
|
|
585
|
-
declare function evaluateHypothesis(manifest: SignedManifest, observed: {
|
|
586
|
-
n: number;
|
|
587
|
-
effect: number;
|
|
588
|
-
pValue: number;
|
|
589
|
-
}): Promise<HypothesisResult>;
|
|
590
|
-
//#endregion
|
|
591
|
-
export { ProportionInterval as A, wilson as B, EProcess as C, eProcess as D, EProcessStep as E, pairedBinaryScale as F, pairedRiskDifference as I, pairedRiskDifferenceExact as L, ScoreRiskDifferenceResult as M, isBinaryOutcomeVector as N, ExactRiskDifferenceResult as O, mcnemar as P, pairedRiskDifferenceScore as R, pairRunRecords as S, EProcessState as T, PairedArmsComparison as _, evaluateHypothesis as a, comparePairedArms as b, signManifest as c, MatchedPair as d, MatchedRunRecordPair as f, PairedArmRow as g, PairRunRecordsResult as h, SignedManifestAlgo as i, RiskDifferenceResult as j, McNemarResult as k, verifyManifest as l, PairArmsResult as m, HypothesisResult as n, hashJson as o, PairArmsOptions as p, SignedManifest as r, manifestContentDigest as s, HypothesisManifest as t, ComparePairedArmsOptions as u, PairedCorrectness as v, EProcessOptions as w, pairArms as x, PairedMetricDelta as y, passAtK as z };
|
|
592
|
-
//# sourceMappingURL=pre-registration-BoI4ucR3.d.ts.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"pre-registration-BoI4ucR3.d.ts","names":[],"sources":["../src/statistics/paired-binary.ts","../src/statistics/sequential-eprocess.ts","../src/paired-arms.ts","../src/pre-registration.ts"],"mappings":";;;;UAgBiB;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;;;;;;;;;;;;;;;;;;;;;;;;;iBA2CzD,sBAAsB,QAAQ;;UAU7B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;iBAkBc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;UAkCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+Bc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;UAyEc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8Dc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;;;;;;;;;;iBA6Ea,kBACd,QAAQ,mBACR,OAAO;;;;;;;;;;;iBA0BO,QAAQ,WAAW,WAAW;;;UCjgB7B;;;EAGf;;;;EAIA;;;;EAIA;;;;;;;;EAQA,SAAS;;UAGM;;EAEf;;EAEA;;EAEA;;;;;;;;;;UAWe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;EAEA;;;EAGA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;UCpDrC;;;EAGf;;;;;EAKA;;EAEA;;EAEA;;EAEA,UAAU;;UAGK;;EAEf;;EAEA;;;UAIe;EACf;;;;EAIA;EACA,UAAU;EACV,WAAW;;UAGI;;EAEf,OAAO;;;EAGP,kBAAkB;;EAElB,mBAAmB;;;;;;;;;;;;;;;;;;;;iBAqBL,SAAS,eAAe,gBAAgB,MAAM,kBAAkB;;UA8G/D;;EAEf;;EAEA;;EAEA,SAAS;;EAET,gBAAgB;;;UAID;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,aAAa;;EAEb;IAAY;IAAW;;;UAGR,iCAAiC;;;;EAIhD;;EAEA,YAAY;;UAGG;EACf;EACA;EACA;;;EAGA,aAAa;EACb,cAAc;;;;;;;;;;;;;;;iBAgBA,kBACd,eAAe,gBACf,MAAM,2BACL;UAmEc;EACf;EACA;EACA,UAAU;EACV,WAAW;;UAGI;EACf,OAAO;EACP,kBAAkB;EAClB,mBAAmB;;;;;;;;;iBAeL,eACd,uBAAuB,aACvB,wBAAwB,cACvB;;;;;;;;;;;;;;;;;;UClWc;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;;;;;;;;;;;;;;KAeU;UAEK,uBAAuB;;EAEtC;;;;;;;;;EASA,OAAO;;UAGQ;EACf,UAAU;EACV;EACA;EACA;;;EAGA;;EAEA,kBAAkB;EAGlB;;;;;;;;;;;;;;;;;iBAkBoB,SAAS,GAAG,KAAK,IAAI;;;;;;;iBA+B3B,sBAAsB,UAAU;;;;;;iBAe1B,aAAa,GAAG,qBAAqB,QAAQ;;;;;iBAS7C,eAAe,GAAG,iBAAiB;;;;;iBAQnC,mBACpB,UAAU,gBACV;EAAY;EAAW;EAAgB;IACtC,QAAQ"}
|