@tangle-network/agent-eval 0.180.0 → 0.181.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +60 -0
- package/README.md +119 -159
- package/dist/adapters/http.d.ts +2 -2
- package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
- package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
- package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
- package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
- package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
- package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +3 -3
- package/dist/ast-CP9ae9B0.js +557 -0
- package/dist/ast-CP9ae9B0.js.map +1 -0
- package/dist/ast-hI-vjW6J.d.ts +457 -0
- package/dist/ast-hI-vjW6J.d.ts.map +1 -0
- package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
- package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
- package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
- package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
- package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
- package/dist/cli.js +4 -7
- package/dist/cli.js.map +1 -1
- package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
- package/dist/client-CXE-U1SA.js.map +1 -0
- package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
- package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +10 -9
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
- package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
- package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
- package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
- package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
- package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
- package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
- package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
- package/dist/engine-DS1cysJy.d.ts.map +1 -0
- package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
- package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
- package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
- package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +27 -477
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +95 -559
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
- package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
- package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
- package/dist/index-Bp_6sj3x.d.ts.map +1 -0
- package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
- package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
- package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
- package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
- package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
- package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
- package/dist/index.d.ts +28 -28
- package/dist/index.js +24 -15
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
- package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
- package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
- package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
- package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
- package/dist/journal-Cs9f7385.js.map +1 -0
- package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
- package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
- package/dist/llm-judge-DEFZeSiu.js.map +1 -0
- package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
- package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +138 -7
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +245 -97
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
- package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/outcome-store-BXlkwMPR.js +131 -0
- package/dist/outcome-store-BXlkwMPR.js.map +1 -0
- package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
- package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
- package/dist/pipelines/index.js +1 -1
- package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
- package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
- package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
- package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
- package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
- package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
- package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
- package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
- package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +3 -3
- package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
- package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
- package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
- package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
- package/dist/rl.d.ts +53 -99
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +182 -169
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
- package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
- package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
- package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
- package/dist/run-record-Br-Yzt_k.js +464 -0
- package/dist/run-record-Br-Yzt_k.js.map +1 -0
- package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
- package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
- package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
- package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
- package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
- package/dist/sequential-DAsyV2T9.js.map +1 -0
- package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
- package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
- package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
- package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
- package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
- package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
- package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
- package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
- package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
- package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
- package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
- package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
- package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
- package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
- package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
- package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
- package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
- package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
- package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
- package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
- package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
- package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/docs/adapters-observability.md +14 -0
- package/docs/campaign-proposers.md +86 -128
- package/docs/charter.md +108 -112
- package/docs/concepts.md +157 -69
- package/docs/design/mlbenchmarks-book-review.md +440 -0
- package/docs/design/mlbenchmarks-review/observations.json +713 -0
- package/docs/design/mlbenchmarks-review/probes.mts +476 -0
- package/docs/design/mlbenchmarks-review/sources.json +200 -0
- package/docs/design/self-improvement-evidence-audit.md +263 -0
- package/docs/design.md +2 -1
- package/docs/eval-surface-map.md +95 -42
- package/docs/evaluation-integrity.md +220 -0
- package/docs/experiment.md +111 -55
- package/docs/feature-guide.md +5 -6
- package/docs/hosted-ingest-spec.md +4 -11
- package/docs/insight-report.md +187 -455
- package/docs/outcome-validity.md +182 -0
- package/docs/product-eval-adoption.md +1 -2
- package/docs/research-report-methodology.md +7 -7
- package/docs/search-history-receipts.md +8 -0
- package/docs/statistical-evidence.md +129 -0
- package/docs/verdicts.md +76 -49
- package/package.json +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
- package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
- package/dist/client-BlLY6o2w.js.map +0 -1
- package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
- package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
- package/dist/engine-CX8ReXkn.d.ts.map +0 -1
- package/dist/index-BxWvILU8.d.ts.map +0 -1
- package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
- package/dist/ledger-core-Cs9f7385.js.map +0 -1
- package/dist/llm-judge-v80Kmu9g.js.map +0 -1
- package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
- package/dist/outcome-store-ChBKlTd_.js +0 -75
- package/dist/outcome-store-ChBKlTd_.js.map +0 -1
- package/dist/promotion-policy-DWOm70gx.js.map +0 -1
- package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
- package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
- package/dist/run-record-CR63CpHK.js +0 -216
- package/dist/run-record-CR63CpHK.js.map +0 -1
- package/dist/sequential-B5gXgcyp.js.map +0 -1
- package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
- package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
|
@@ -1,29 +1,16 @@
|
|
|
1
1
|
import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
|
|
2
|
-
import {
|
|
2
|
+
import { s as decidePairedPromotion } from "./run-record-Br-Yzt_k.js";
|
|
3
|
+
import { at as pairHoldout, nt as detectScale } from "./campaign-evidence-B8oF9xQ6.js";
|
|
3
4
|
//#region src/campaign/gates/promotion-policy.ts
|
|
4
5
|
/**
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
6
|
+
* Build paired evidence for each objective and apply a promotion policy.
|
|
7
|
+
* The default policy requires at least one gain and every regression floor to
|
|
8
|
+
* clear. A floor can fail because a larger loss remains plausible; that does
|
|
9
|
+
* not demonstrate an observed regression. Missing evidence stays unresolved.
|
|
10
|
+
* Confidence intervals apply to each axis, without a multiplicity adjustment.
|
|
10
11
|
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
* paretoPolicy(ev) // the default strategy
|
|
14
|
-
* paretoSignificanceGate(options): Gate // bus + policy as a Gate
|
|
15
|
-
*
|
|
16
|
-
* The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a
|
|
17
|
-
* potential gain source AND a safety floor (unlike `defaultProductionGate`,
|
|
18
|
-
* where only `composite` can win and `criticalDimensions` are pure floors). A
|
|
19
|
-
* candidate ships iff it weakly DOMINATES the baseline at the confidence level —
|
|
20
|
-
* no objective credibly worse (CI floor breach) AND at least one objective
|
|
21
|
-
* credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work
|
|
22
|
-
* (NOT folded into hold: "gather more reps" and "reject" are different actions).
|
|
23
|
-
*
|
|
24
|
-
* Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate
|
|
25
|
-
* per-side cost, no per-cell observation vector to bootstrap. Treat them as hard
|
|
26
|
-
* constraints (compose with a budget gate via `composeGate`), not faked CIs.
|
|
12
|
+
* Cost and latency remain aggregate constraints because GateContext does not
|
|
13
|
+
* supply paired observations for them. Compose a budget gate when needed.
|
|
27
14
|
*/
|
|
28
15
|
/**
|
|
29
16
|
* The Evidence Bus. For each objective, pair candidate vs baseline by full
|
|
@@ -50,7 +37,8 @@ function buildEvidenceVector(ctx, objectives, opts = {}) {
|
|
|
50
37
|
const before = obj.direction === "maximize" ? paired.before : paired.after;
|
|
51
38
|
const after = obj.direction === "maximize" ? paired.after : paired.before;
|
|
52
39
|
const n = paired.before.length;
|
|
53
|
-
const
|
|
40
|
+
const binaryScale = obj.binaryScale;
|
|
41
|
+
const floorTolerance = obj.floorTolerance ?? .05 * (binaryScale ?? detectScale([...paired.before, ...paired.after]));
|
|
54
42
|
const gainThreshold = obj.gainThreshold ?? 0;
|
|
55
43
|
const bootstrapStatistic = opts.statistic ?? "mean";
|
|
56
44
|
const improvement = decidePairedPromotion(before, after, {
|
|
@@ -59,7 +47,8 @@ function buildEvidenceVector(ctx, objectives, opts = {}) {
|
|
|
59
47
|
statistic: bootstrapStatistic,
|
|
60
48
|
seed,
|
|
61
49
|
threshold: gainThreshold,
|
|
62
|
-
minPairs: opts.minProductiveRuns
|
|
50
|
+
minPairs: opts.minProductiveRuns,
|
|
51
|
+
binaryScale
|
|
63
52
|
});
|
|
64
53
|
const regression = decidePairedPromotion(after, before, {
|
|
65
54
|
confidence,
|
|
@@ -67,7 +56,8 @@ function buildEvidenceVector(ctx, objectives, opts = {}) {
|
|
|
67
56
|
statistic: bootstrapStatistic,
|
|
68
57
|
seed,
|
|
69
58
|
threshold: floorTolerance,
|
|
70
|
-
minPairs: opts.minProductiveRuns
|
|
59
|
+
minPairs: opts.minProductiveRuns,
|
|
60
|
+
binaryScale
|
|
71
61
|
});
|
|
72
62
|
const bootstrap = improvement.bootstrap ?? pairedBootstrap(before, after, {
|
|
73
63
|
confidence,
|
|
@@ -75,8 +65,8 @@ function buildEvidenceVector(ctx, objectives, opts = {}) {
|
|
|
75
65
|
statistic: bootstrapStatistic,
|
|
76
66
|
seed
|
|
77
67
|
});
|
|
78
|
-
const floorBreached =
|
|
79
|
-
const verdict = !improvement.sufficient ? "few_runs" : floorBreached ? "regressed" : improvement.promote ? "improved" : "flat";
|
|
68
|
+
const floorBreached = improvement.low < -floorTolerance || regression.promote;
|
|
69
|
+
const verdict = !improvement.sufficient ? "few_runs" : improvement.indeterminate ? "indeterminate" : floorBreached ? "regressed" : improvement.promote ? "improved" : "flat";
|
|
80
70
|
axes.push({
|
|
81
71
|
name: obj.name,
|
|
82
72
|
source: obj.source,
|
|
@@ -109,16 +99,14 @@ function buildEvidenceVector(ctx, objectives, opts = {}) {
|
|
|
109
99
|
};
|
|
110
100
|
}
|
|
111
101
|
/**
|
|
112
|
-
*
|
|
113
|
-
*
|
|
114
|
-
*
|
|
115
|
-
* (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
|
|
116
|
-
* need_more_work. Statistically equivalent → hold (never ship noise).
|
|
102
|
+
* Require a supported gain and every configured regression floor to clear.
|
|
103
|
+
* A failed floor holds the candidate, including when uncertainty permits a loss.
|
|
104
|
+
* Missing or indeterminate evidence requires more work; no gain holds release.
|
|
117
105
|
*/
|
|
118
106
|
const paretoPolicy = (ev) => {
|
|
119
107
|
const contributingGates = ev.axes.map((ax) => ({
|
|
120
108
|
name: `objective:${ax.name}`,
|
|
121
|
-
status: ax.verdict === "regressed" ? "fail" : ax.verdict === "few_runs" ? "not_evaluated" : "pass",
|
|
109
|
+
status: ax.verdict === "regressed" ? "fail" : ax.verdict === "few_runs" || ax.verdict === "indeterminate" ? "not_evaluated" : "pass",
|
|
122
110
|
detail: {
|
|
123
111
|
direction: ax.direction,
|
|
124
112
|
source: ax.source,
|
|
@@ -139,22 +127,22 @@ const paretoPolicy = (ev) => {
|
|
|
139
127
|
}
|
|
140
128
|
}));
|
|
141
129
|
const regressed = ev.axes.filter((a) => a.verdict === "regressed");
|
|
142
|
-
const
|
|
130
|
+
const insufficient = ev.axes.filter((a) => a.verdict === "few_runs" || a.verdict === "indeterminate");
|
|
143
131
|
const improved = ev.axes.filter((a) => a.verdict === "improved");
|
|
144
132
|
let decision;
|
|
145
133
|
const reasons = [];
|
|
146
134
|
if (regressed.length > 0) {
|
|
147
135
|
decision = "hold";
|
|
148
|
-
for (const a of regressed) reasons.push(`objective '${a.name}'
|
|
149
|
-
} else if (
|
|
136
|
+
for (const a of regressed) reasons.push(`objective '${a.name}' did not clear its regression floor -${a.floorTolerance}: good-direction CI [${a.ci.low.toFixed(3)}, ${a.ci.high.toFixed(3)}] (n=${a.n})`);
|
|
137
|
+
} else if (insufficient.length > 0) {
|
|
150
138
|
decision = "need_more_work";
|
|
151
|
-
for (const a of
|
|
139
|
+
for (const a of insufficient) reasons.push(a.verdict === "few_runs" ? `objective '${a.name}' has only n=${a.n} paired runs — insufficient evidence to claim significance` : `objective '${a.name}' has an indeterminate deciding CI [${a.ci.low}, ${a.ci.high}] — insufficient evidence to clear its regression floor (n=${a.n})`);
|
|
152
140
|
} else if (improved.length > 0) {
|
|
153
141
|
decision = "ship";
|
|
154
|
-
reasons.push(`
|
|
142
|
+
reasons.push(`Supported objective gain: ${improved.map((a) => `'${a.name}' +${a.ci.low > 0 ? a.ci.low.toFixed(3) : a.bootstrap.mean.toFixed(3)} (CI.low ${a.ci.low.toFixed(3)})`).join(", ")}; every regression floor cleared`);
|
|
155
143
|
} else {
|
|
156
144
|
decision = "hold";
|
|
157
|
-
reasons.push("no Pareto improvement:
|
|
145
|
+
reasons.push("no Pareto improvement: no objective shows a significant gain; every regression floor cleared");
|
|
158
146
|
}
|
|
159
147
|
const composite = ev.axes.find((a) => a.source.kind === "composite") ?? ev.axes[0];
|
|
160
148
|
return {
|
|
@@ -183,4 +171,4 @@ function paretoSignificanceGate(options) {
|
|
|
183
171
|
//#endregion
|
|
184
172
|
export { paretoPolicy as n, paretoSignificanceGate as r, buildEvidenceVector as t };
|
|
185
173
|
|
|
186
|
-
//# sourceMappingURL=promotion-policy-
|
|
174
|
+
//# sourceMappingURL=promotion-policy-CDMMxzb6.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"promotion-policy-CDMMxzb6.js","names":[],"sources":["../src/campaign/gates/promotion-policy.ts"],"sourcesContent":["/**\n * Build paired evidence for each objective and apply a promotion policy.\n * The default policy requires at least one gain and every regression floor to\n * clear. A floor can fail because a larger loss remains plausible; that does\n * not demonstrate an observed regression. Missing evidence stays unresolved.\n * Confidence intervals apply to each axis, without a multiplicity adjustment.\n *\n * Cost and latency remain aggregate constraints because GateContext does not\n * supply paired observations for them. Compose a budget gate when needed.\n */\n\nimport {\n decidePairedPromotion,\n type PairedDecisionMethod,\n type PairedDecisionStatistic,\n type PairedMcNemarEvidence,\n} from '../../paired-promotion-decision'\nimport type { Direction } from '../../pareto'\nimport {\n DECISION_PAIRED_DELTA_STATISTIC,\n type PairedBootstrapResult,\n pairedBootstrap,\n} from '../../statistics'\nimport type { Gate, GateContext, GateDecision, GateResult, JudgeScore, Scenario } from '../types'\nimport { detectScale, pairHoldout } from './statistical-heldout'\n\n/** Where an objective's per-cell scalar comes from. `composite` reads the\n * judge's composite; `dimension` reads a named per-dimension score. */\nexport type ObjectiveSource = { kind: 'composite' } | { kind: 'dimension'; dimension: string }\n\nexport interface PromotionObjective {\n /** Stable label used in reports + `contributingGates`. */\n name: string\n source: ObjectiveSource\n /** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients\n * the paired delta so a positive bootstrap always means \"candidate better\". */\n direction: Direction\n /** Declared binary support {0, binaryScale}, including zero-only observations.\n * Must be finite and positive; paired cell scores must be 0 or this scale.\n * Uses the risk-difference mean and rejects the 'median' statistic. */\n binaryScale?: number\n /** The good-direction paired-delta CI lower bound must EXCEED this to count\n * as a significant gain on this axis. Interpreted in the judge's native\n * scale. Default 0 (⇒ \"confidently better\"). */\n gainThreshold?: number\n /** A floor breach (regression) is declared when the good-direction CI lower\n * bound is below −floorTolerance, or when the exact small-sample test proves\n * a drop past it. Defaults to 0.05 times the declared binary scale, or\n * auto-scales off observed magnitudes (0.05 on [0,1], 5 on 0-100). */\n floorTolerance?: number\n}\n\n/** Per-axis verdict from the shared paired decision rule.\n * 'regressed' includes uncertainty that prevents clearing the regression floor. */\nexport type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs' | 'indeterminate'\n\nexport interface AxisEvidence {\n name: string\n source: ObjectiveSource\n direction: Direction\n /** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):\n * a positive value means the candidate is better on this axis.\n *\n * DIAGNOSTIC on a pass/fail axis: there the verdict is decided on Tango's\n * score interval instead, because a percentile bootstrap over a three-atom\n * delta lattice is not a valid interval at the nonzero margin `floorTolerance`\n * and `gainThreshold` create. `ci` carries the interval that decided. */\n bootstrap: PairedBootstrapResult\n /** Which paired statistic `bootstrap.low`/`.high` bracket. `'mean'` unless the\n * caller asked for the median — on a pass/fail axis the median and its whole\n * CI are pinned at 0 by tie domination and can see neither a gain nor a\n * regression. `bootstrap.median` still carries the median point estimate. */\n bootstrapStatistic: 'median' | 'mean'\n /** The interval the axis verdict was actually decided on, good-direction and\n * in the axis's native units. */\n ci: { low: number; high: number }\n /** Which estimator produced `ci`. */\n decisionStatistic: PairedDecisionStatistic\n /** McNemar's exact evidence on a pass/fail axis; null otherwise. */\n mcnemar: PairedMcNemarEvidence | null\n /** `ci` has zero width or non-finite bounds. It cannot establish a gain or\n * clear a regression floor, regardless of the point estimate. */\n indeterminate: boolean\n /** Paired observations contributing to this axis. */\n n: number\n minimumRequired: number\n decisionMethod: PairedDecisionMethod\n gainThreshold: number\n floorTolerance: number\n verdict: AxisVerdict\n}\n\nexport interface EvidenceVector {\n /** One entry per objective — NOTHING averaged across axes. */\n axes: AxisEvidence[]\n /** Smallest paired n across axes that produced observations — the binding\n * evidence-sufficiency constraint. 0 when no axis produced observations. */\n minN: number\n /** Aggregate per-side cost from the gate context (a constraint input, not a\n * CI axis — see the module header). */\n cost: { candidate: number; baseline: number }\n}\n\n/** A promotion strategy: a pure function from the evidence vector to a verdict.\n * Many policies can run over the same `EvidenceVector` and disagree — that's\n * the point (competing strategies, shared evidence). */\nexport type PromotionPolicy = (ev: EvidenceVector) => GateResult\n\nexport interface BuildEvidenceVectorOptions {\n /** Minimum paired observations before an axis can claim significance; below\n * it the axis is `few_runs`. The exact small-sample test may require more\n * observations at the selected confidence. */\n minProductiveRuns?: number\n /** Confidence level for every axis bootstrap. Default 0.95. */\n confidence?: number\n /** Bootstrap resamples. Default 2000. */\n resamples?: number\n /** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */\n seed?: number\n /** Paired statistic every axis CI is computed on. Default `'mean'` — see\n * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */\n statistic?: 'mean' | 'median'\n}\n\n/**\n * The Evidence Bus. For each objective, pair candidate vs baseline by full\n * cellId and bootstrap a CI on the good-direction paired delta. Reuses the\n * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so\n * a single source of truth governs pairing granularity + scale handling.\n */\nexport function buildEvidenceVector<TArtifact, TScenario extends Scenario>(\n ctx: GateContext<TArtifact, TScenario>,\n objectives: PromotionObjective[],\n opts: BuildEvidenceVectorOptions = {},\n): EvidenceVector {\n if (objectives.length === 0) {\n throw new Error('buildEvidenceVector: at least 1 objective required')\n }\n const confidence = opts.confidence ?? 0.95\n const resamples = opts.resamples ?? 2000\n const seed = opts.seed ?? 1337\n const baseline = ctx.baselineJudgeScores ?? ctx.judgeScores\n const scenarioIds = new Set(ctx.scenarios.map((s) => s.id))\n\n const axes: AxisEvidence[] = []\n for (const obj of objectives) {\n let select: (s: JudgeScore) => number | undefined\n if (obj.source.kind === 'composite') {\n select = (s) => s.composite\n } else {\n const dim = obj.source.dimension\n select = (s) => s.dimensions[dim]\n }\n const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select)\n // Orient to the good direction: maximize ⇒ bootstrap (candidate − baseline);\n // minimize ⇒ bootstrap (baseline − candidate) by swapping args, so a\n // positive bootstrap always reads as \"candidate better on this axis\".\n const before = obj.direction === 'maximize' ? paired.before : paired.after\n const after = obj.direction === 'maximize' ? paired.after : paired.before\n const n = paired.before.length\n const binaryScale = obj.binaryScale\n const floorTolerance =\n obj.floorTolerance ?? 0.05 * (binaryScale ?? detectScale([...paired.before, ...paired.after]))\n const gainThreshold = obj.gainThreshold ?? 0\n // Axes are decided on the MEAN paired delta — which for a pass/fail axis is\n // exactly the change in success rate. The median is structurally blind on\n // the shapes eval data lands in: with most pairs tied its bootstrap CI\n // collapses to [0,0] and the axis reads 'flat', hiding real gains AND real\n // regressions — pass/fail axes on ANY encoding ({0,1} and the 0-100 one\n // `detectScale` exists for), and low-cardinality axes even below half ties.\n // Orthogonal to `pairedDeltaTest`'s own small-sample switch: that picks the\n // TEST (bootstrap CI at n ≥ 20, exact sign test below), this picks the\n // ESTIMATOR the test is applied to. Both are needed — an exact sign test on\n // a tie-pinned median is still blind.\n const bootstrapStatistic = opts.statistic ?? DECISION_PAIRED_DELTA_STATISTIC\n // Both burdens of proof route through the ONE shared rule\n // (`decidePairedPromotion`), so a pass/fail axis is judged on Tango's score\n // interval — the only paired-binary construction valid at the nonzero\n // margins `gainThreshold` / `floorTolerance` create — and a zero-width\n // interval cannot buy a verdict in either direction.\n const improvement = decidePairedPromotion(before, after, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n threshold: gainThreshold,\n minPairs: opts.minProductiveRuns,\n binaryScale,\n })\n const regression = decidePairedPromotion(after, before, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n threshold: floorTolerance,\n minPairs: opts.minProductiveRuns,\n binaryScale,\n })\n const bootstrap =\n improvement.bootstrap ??\n pairedBootstrap(before, after, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n })\n // A floor breach fires on EITHER burden of proof, because they cover\n // different failures and the floor is the anti-Goodhart guard:\n // - `improvement.low < -floorTolerance` — the CREDIBLE WORST CASE exceeds\n // the tolerance. This is the contract `AxisEvidence.floorTolerance` and\n // `paretoPolicy`'s own reason string state, and it is the conservative\n // posture a safety axis needs: block unless the data can rule the\n // breach out, rather than waiting for the breach to be proven. Read off\n // the DECIDING interval, so a pass/fail axis is not screened by a\n // bootstrap that is pinned wherever ties dominate.\n // - `regression.promote` — a PROVEN drop past the tolerance. Adds the\n // small-sample path, where the decision is an exact sign test because\n // the bootstrap interval is descriptive only.\n // A tied binary axis still has uncertainty about unseen discordant pairs.\n // Its diagnostic bootstrap collapses to zero, so only the deciding score\n // interval can establish that a regression stays within the declared floor.\n const floorBreached = improvement.low < -floorTolerance || regression.promote\n // Floor check precedes the gain check: a credible regression must never be\n // masked as \"improved\". With the defaults (gainThreshold 0, positive floor)\n // the regions are disjoint and order is moot, but a consumer who sets a\n // negative gainThreshold (\"accept small dips\") could otherwise have a real\n // floor breach classified as a gain — anti-Goodhart wins the tie.\n const verdict: AxisVerdict = !improvement.sufficient\n ? 'few_runs'\n : improvement.indeterminate\n ? 'indeterminate'\n : floorBreached\n ? 'regressed'\n : improvement.promote\n ? 'improved'\n : 'flat'\n axes.push({\n name: obj.name,\n source: obj.source,\n direction: obj.direction,\n bootstrap,\n bootstrapStatistic,\n ci: { low: improvement.low, high: improvement.high },\n decisionStatistic: improvement.statistic,\n mcnemar: improvement.mcnemar,\n indeterminate: improvement.indeterminate,\n n,\n minimumRequired: improvement.minimumPairs,\n decisionMethod: improvement.method,\n gainThreshold,\n floorTolerance,\n verdict,\n })\n }\n const ns = axes.map((a) => a.n).filter((n) => n > 0)\n const minN = ns.length > 0 ? Math.min(...ns) : 0\n return { axes, minN, cost: { candidate: ctx.cost.candidate, baseline: ctx.cost.baseline } }\n}\n\n/**\n * Require a supported gain and every configured regression floor to clear.\n * A failed floor holds the candidate, including when uncertainty permits a loss.\n * Missing or indeterminate evidence requires more work; no gain holds release.\n */\nexport const paretoPolicy: PromotionPolicy = (ev) => {\n const contributingGates = ev.axes.map((ax) => ({\n name: `objective:${ax.name}`,\n status:\n ax.verdict === 'regressed'\n ? ('fail' as const)\n : ax.verdict === 'few_runs' || ax.verdict === 'indeterminate'\n ? ('not_evaluated' as const)\n : ('pass' as const),\n detail: {\n direction: ax.direction,\n source: ax.source,\n verdict: ax.verdict,\n n: ax.n,\n deltaMedian: ax.bootstrap.median,\n ciLow: ax.ci.low,\n ciHigh: ax.ci.high,\n decisionStatistic: ax.decisionStatistic,\n decisionMethod: ax.decisionMethod,\n mcnemar: ax.mcnemar,\n indeterminate: ax.indeterminate,\n bootstrapCiLow: ax.bootstrap.low,\n bootstrapCiHigh: ax.bootstrap.high,\n confidence: ax.bootstrap.confidence,\n gainThreshold: ax.gainThreshold,\n floorTolerance: ax.floorTolerance,\n },\n }))\n\n const regressed = ev.axes.filter((a) => a.verdict === 'regressed')\n const insufficient = ev.axes.filter(\n (a) => a.verdict === 'few_runs' || a.verdict === 'indeterminate',\n )\n const improved = ev.axes.filter((a) => a.verdict === 'improved')\n\n let decision: GateDecision\n const reasons: string[] = []\n if (regressed.length > 0) {\n // A gain on another axis cannot excuse an unresolved regression floor.\n decision = 'hold'\n for (const a of regressed) {\n reasons.push(\n `objective '${a.name}' did not clear its regression floor -${a.floorTolerance}: good-direction CI [${a.ci.low.toFixed(3)}, ${a.ci.high.toFixed(3)}] (n=${a.n})`,\n )\n }\n } else if (insufficient.length > 0) {\n // An unresolved axis cannot establish either a gain or a safe floor.\n decision = 'need_more_work'\n for (const a of insufficient) {\n reasons.push(\n a.verdict === 'few_runs'\n ? `objective '${a.name}' has only n=${a.n} paired runs — insufficient evidence to claim significance`\n : `objective '${a.name}' has an indeterminate deciding CI [${a.ci.low}, ${a.ci.high}] — insufficient evidence to clear its regression floor (n=${a.n})`,\n )\n }\n } else if (improved.length > 0) {\n decision = 'ship'\n reasons.push(\n `Supported objective gain: ${improved\n .map(\n (a) =>\n `'${a.name}' +${a.ci.low > 0 ? a.ci.low.toFixed(3) : a.bootstrap.mean.toFixed(3)} (CI.low ${a.ci.low.toFixed(3)})`,\n )\n .join(', ')}; every regression floor cleared`,\n )\n } else {\n // Every floor cleared, but no objective demonstrated a significant gain.\n decision = 'hold'\n reasons.push(\n 'no Pareto improvement: no objective shows a significant gain; every regression floor cleared',\n )\n }\n\n // `delta` surfaces the composite axis if present, else the first axis — a\n // single convenience scalar; the vector lives in `contributingGates`.\n const composite = ev.axes.find((a) => a.source.kind === 'composite') ?? ev.axes[0]\n return { decision, reasons, contributingGates, delta: composite?.bootstrap.median }\n}\n\nexport interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {\n /** The objective vector. Every axis is both a gain source and a safety floor. */\n objectives: PromotionObjective[]\n /** Strategy applied to the evidence vector. Default `paretoPolicy`. Override\n * to run a stricter/looser strategy over the SAME bus (competing policies). */\n policy?: PromotionPolicy\n /** Override the gate name in reports. */\n name?: string\n}\n\n/**\n * Wrap the bus + a policy as a `Gate`. Plugs into the existing\n * `runImprovementLoop({ gate })` slot and composes via `composeGate`; default\n * loop behavior is unchanged because consumers opt in by passing this gate.\n */\nexport function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(\n options: ParetoSignificanceGateOptions,\n): Gate<TArtifact, TScenario> {\n if (options.objectives.length === 0) {\n throw new Error('paretoSignificanceGate: at least 1 objective required')\n }\n const policy = options.policy ?? paretoPolicy\n return {\n name: options.name ?? 'paretoSignificanceGate',\n async decide(ctx: GateContext<TArtifact, TScenario>): Promise<GateResult> {\n const ev = buildEvidenceVector(ctx, options.objectives, options)\n return policy(ev)\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;AAkIA,SAAgB,oBACd,KACA,YACA,OAAmC,CAAC,GACpB;CAChB,IAAI,WAAW,WAAW,GACxB,MAAM,IAAI,MAAM,oDAAoD;CAEtE,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,WAAW,IAAI,uBAAuB,IAAI;CAChD,MAAM,cAAc,IAAI,IAAI,IAAI,UAAU,KAAK,MAAM,EAAE,EAAE,CAAC;CAE1D,MAAM,OAAuB,CAAC;CAC9B,KAAK,MAAM,OAAO,YAAY;EAC5B,IAAI;EACJ,IAAI,IAAI,OAAO,SAAS,aACtB,UAAU,MAAM,EAAE;OACb;GACL,MAAM,MAAM,IAAI,OAAO;GACvB,UAAU,MAAM,EAAE,WAAW;EAC/B;EACA,MAAM,SAAS,YAAY,IAAI,aAAa,UAAU,aAAa,MAAM;EAIzE,MAAM,SAAS,IAAI,cAAc,aAAa,OAAO,SAAS,OAAO;EACrE,MAAM,QAAQ,IAAI,cAAc,aAAa,OAAO,QAAQ,OAAO;EACnE,MAAM,IAAI,OAAO,OAAO;EACxB,MAAM,cAAc,IAAI;EACxB,MAAM,iBACJ,IAAI,kBAAkB,OAAQ,eAAe,YAAY,CAAC,GAAG,OAAO,QAAQ,GAAG,OAAO,KAAK,CAAC;EAC9F,MAAM,gBAAgB,IAAI,iBAAiB;EAW3C,MAAM,qBAAqB,KAAK,aAAA;EAMhC,MAAM,cAAc,sBAAsB,QAAQ,OAAO;GACvD;GACA;GACA,WAAW;GACX;GACA,WAAW;GACX,UAAU,KAAK;GACf;EACF,CAAC;EACD,MAAM,aAAa,sBAAsB,OAAO,QAAQ;GACtD;GACA;GACA,WAAW;GACX;GACA,WAAW;GACX,UAAU,KAAK;GACf;EACF,CAAC;EACD,MAAM,YACJ,YAAY,aACZ,gBAAgB,QAAQ,OAAO;GAC7B;GACA;GACA,WAAW;GACX;EACF,CAAC;EAgBH,MAAM,gBAAgB,YAAY,MAAM,CAAC,kBAAkB,WAAW;EAMtE,MAAM,UAAuB,CAAC,YAAY,aACtC,aACA,YAAY,gBACV,kBACA,gBACE,cACA,YAAY,UACV,aACA;EACV,KAAK,KAAK;GACR,MAAM,IAAI;GACV,QAAQ,IAAI;GACZ,WAAW,IAAI;GACf;GACA;GACA,IAAI;IAAE,KAAK,YAAY;IAAK,MAAM,YAAY;GAAK;GACnD,mBAAmB,YAAY;GAC/B,SAAS,YAAY;GACrB,eAAe,YAAY;GAC3B;GACA,iBAAiB,YAAY;GAC7B,gBAAgB,YAAY;GAC5B;GACA;GACA;EACF,CAAC;CACH;CACA,MAAM,KAAK,KAAK,KAAK,MAAM,EAAE,CAAC,CAAC,CAAC,QAAQ,MAAM,IAAI,CAAC;CAEnD,OAAO;EAAE;EAAM,MADF,GAAG,SAAS,IAAI,KAAK,IAAI,GAAG,EAAE,IAAI;EAC1B,MAAM;GAAE,WAAW,IAAI,KAAK;GAAW,UAAU,IAAI,KAAK;EAAS;CAAE;AAC5F;;;;;;AAOA,MAAa,gBAAiC,OAAO;CACnD,MAAM,oBAAoB,GAAG,KAAK,KAAK,QAAQ;EAC7C,MAAM,aAAa,GAAG;EACtB,QACE,GAAG,YAAY,cACV,SACD,GAAG,YAAY,cAAc,GAAG,YAAY,kBACzC,kBACA;EACT,QAAQ;GACN,WAAW,GAAG;GACd,QAAQ,GAAG;GACX,SAAS,GAAG;GACZ,GAAG,GAAG;GACN,aAAa,GAAG,UAAU;GAC1B,OAAO,GAAG,GAAG;GACb,QAAQ,GAAG,GAAG;GACd,mBAAmB,GAAG;GACtB,gBAAgB,GAAG;GACnB,SAAS,GAAG;GACZ,eAAe,GAAG;GAClB,gBAAgB,GAAG,UAAU;GAC7B,iBAAiB,GAAG,UAAU;GAC9B,YAAY,GAAG,UAAU;GACzB,eAAe,GAAG;GAClB,gBAAgB,GAAG;EACrB;CACF,EAAE;CAEF,MAAM,YAAY,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,WAAW;CACjE,MAAM,eAAe,GAAG,KAAK,QAC1B,MAAM,EAAE,YAAY,cAAc,EAAE,YAAY,eACnD;CACA,MAAM,WAAW,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,UAAU;CAE/D,IAAI;CACJ,MAAM,UAAoB,CAAC;CAC3B,IAAI,UAAU,SAAS,GAAG;EAExB,WAAW;EACX,KAAK,MAAM,KAAK,WACd,QAAQ,KACN,cAAc,EAAE,KAAK,wCAAwC,EAAE,eAAe,uBAAuB,EAAE,GAAG,IAAI,QAAQ,CAAC,EAAE,IAAI,EAAE,GAAG,KAAK,QAAQ,CAAC,EAAE,OAAO,EAAE,EAAE,EAC/J;CAEJ,OAAO,IAAI,aAAa,SAAS,GAAG;EAElC,WAAW;EACX,KAAK,MAAM,KAAK,cACd,QAAQ,KACN,EAAE,YAAY,aACV,cAAc,EAAE,KAAK,eAAe,EAAE,EAAE,8DACxC,cAAc,EAAE,KAAK,sCAAsC,EAAE,GAAG,IAAI,IAAI,EAAE,GAAG,KAAK,6DAA6D,EAAE,EAAE,EACzJ;CAEJ,OAAO,IAAI,SAAS,SAAS,GAAG;EAC9B,WAAW;EACX,QAAQ,KACN,6BAA6B,SAC1B,KACE,MACC,IAAI,EAAE,KAAK,KAAK,EAAE,GAAG,MAAM,IAAI,EAAE,GAAG,IAAI,QAAQ,CAAC,IAAI,EAAE,UAAU,KAAK,QAAQ,CAAC,EAAE,WAAW,EAAE,GAAG,IAAI,QAAQ,CAAC,EAAE,EACpH,CAAC,CACA,KAAK,IAAI,EAAE,iCAChB;CACF,OAAO;EAEL,WAAW;EACX,QAAQ,KACN,8FACF;CACF;CAIA,MAAM,YAAY,GAAG,KAAK,MAAM,MAAM,EAAE,OAAO,SAAS,WAAW,KAAK,GAAG,KAAK;CAChF,OAAO;EAAE;EAAU;EAAS;EAAmB,OAAO,WAAW,UAAU;CAAO;AACpF;;;;;;AAiBA,SAAgB,uBACd,SAC4B;CAC5B,IAAI,QAAQ,WAAW,WAAW,GAChC,MAAM,IAAI,MAAM,uDAAuD;CAEzE,MAAM,SAAS,QAAQ,UAAU;CACjC,OAAO;EACL,MAAM,QAAQ,QAAQ;EACtB,MAAM,OAAO,KAA6D;GACxE,MAAM,KAAK,oBAAoB,KAAK,QAAQ,YAAY,OAAO;GAC/D,OAAO,OAAO,EAAE;EAClB;CACF;AACF"}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { c as CostLedgerHandle } from "./cost-ledger-DbQdN3nO.js";
|
|
2
|
-
import { c as AnalystRunInputs, i as AnalystFinding, l as AnalystRunResult, n as AnalystContext, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary } from "./types-
|
|
3
|
-
import {
|
|
4
|
-
import { i as ExactAnalystRunEvent, o as ExactAnalystRunResult, u as ExactExecutionComponentIdentity } from "./exact-types-
|
|
2
|
+
import { c as AnalystRunInputs, i as AnalystFinding, l as AnalystRunResult, n as AnalystContext, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary } from "./types-D7gEdPoQ.js";
|
|
3
|
+
import { x as ChatClient } from "./types-CBbLtr2J.js";
|
|
4
|
+
import { i as ExactAnalystRunEvent, o as ExactAnalystRunResult, u as ExactExecutionComponentIdentity } from "./exact-types-BZDe0W2D.js";
|
|
5
5
|
//#region src/analyst/registry.d.ts
|
|
6
6
|
interface AnalystHooks {
|
|
7
7
|
/** Legacy runs may mutate ctx; exact runs provide a frozen observational context. */
|
|
@@ -173,4 +173,4 @@ declare class AnalystRegistry {
|
|
|
173
173
|
}
|
|
174
174
|
//#endregion
|
|
175
175
|
export { ExactAnalystBudgetPolicy as a, RegistryRunOpts as c, BudgetPolicy as i, AnalystRegistry as n, ExactAnalystRunExecutionError as o, AnalystRegistryOptions as r, ExactRegistryRunOpts as s, AnalystHooks as t };
|
|
176
|
-
//# sourceMappingURL=registry-
|
|
176
|
+
//# sourceMappingURL=registry-BRbB6Y0v.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"registry-
|
|
1
|
+
{"version":3,"file":"registry-BRbB6Y0v.d.ts","names":[],"sources":["../src/analyst/registry.ts"],"mappings":";;;;;UAqDiB;;EAEf,iBAAiB;IACf,SAAS;IACT,KAAK;IACL;aACS;;EAEX,gBAAgB;IACd,SAAS;IACT,SAAS;IACT,UAAU;IACV;aACS;;;;;;EAMX,SAAS;IACP,SAAS;IACT,OAAO;IACP;MACE,+BAA+B,QAAQ;;EAE3C,YAAY;IAAQ,QAAQ;aAA4B;;UAGzC;;EAEf;;EAEA,UAAU;;;;;;;EAOV,YAAY;IACV,SAAS;IACT;IACA;IACA;;;UAIa;;EAEf,OAAO;;EAEP,OAAO,aAAa,SAAS;;EAE7B,QAAQ;;EAER,gBAAgB;;EAEhB,gBAAgB;;EAEhB,eAAe;;UAGA;;EAEf;;EAEA;;EAEA,SAAS;;EAET;;EAEA,SAAS;;EAET,aAAa;;EAEb;;EAEA,OAAO;;;;;;;;;EASP,gBAAgB,cAAc,kBAAkB,eAAe,cAAc;;;;;;EAM7E;;;KAIU;WAEG;WACA;;WAGA;WACA;;WAEA,SAAS,SAAS;;;;;;;;;;UAWhB;WACN;WACA,QAAQ;WACR;WACA,QAAQ;WACR,YAAY;WACZ,oBAAoB;WACpB;WACA,MAAM,SAAS;WACf,eACL,cAAc,kBACd,SAAS,eAAe,cAAc;WAEjC;WACA;WACA;WACA;;;cAqCE,sCAAsC;WACxC;WACA,QAAQ;EAEjB,YAAY,iBAAiB,QAAQ,uBAAuB,UAAU;;cAU3D;mBACM;mBACA;EAEjB,YAAY,UAAS;EAIrB,SAAS,SAAS;EAsBlB,QAAQ;IACN;IACA;IACA;IACA,MAAM;;EAUF,IACJ,eACA,QAAQ,kBACR,UAAS,kBACR,QAAQ;;EAUL,SACJ,eACA,QAAQ,kBACR,SAAS,uBACR,QAAQ;;EAQJ,eACL,eACA,QAAQ,kBACR,SAAS,uBACR,eAAe;;;;;;;;;;;;EAmBX,UACL,eACA,QAAQ,kBACR,UAAS,kBACR,eAAe;UAIV;UA8BA;UAwFO;UA2aP;UAaA;UAUA"}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { o as FailureClass } from "./schema-CR5cpjQ3.js";
|
|
2
|
-
import { a as RunRecord, s as RunSplitTag } from "./run-record-
|
|
2
|
+
import { a as RunRecord, s as RunSplitTag } from "./run-record-BiTWauyO.js";
|
|
3
3
|
import { i as DatasetSplit, n as DatasetManifest, r as DatasetScenario } from "./dataset-DQqhOCPt.js";
|
|
4
|
-
import { b as GateDecision } from "./summary-report-
|
|
4
|
+
import { b as GateDecision } from "./summary-report-D1h4dlrK.js";
|
|
5
5
|
//#region src/statistics/rank-tests.d.ts
|
|
6
6
|
/** How a rank test's p-value was actually computed. */
|
|
7
7
|
type RankTestMethod = 'exact' | 'permutation' | 'asymptotic';
|
|
@@ -310,4 +310,4 @@ declare function evaluateReleaseConfidence(input: ReleaseConfidenceInput): Relea
|
|
|
310
310
|
declare function assertReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard;
|
|
311
311
|
//#endregion
|
|
312
312
|
export { RankTestMethod as C, WilcoxonSignedRankResult as D, WILCOXON_EXACT_MAX_N as E, mannWhitneyU as O, MannWhitneyResult as S, RankTestOptions as T, bootstrapCi as _, ReleaseConfidenceIssue as a, MANN_WHITNEY_EXACT_MAX_STATES as b, ReleaseConfidenceStatus as c, assertReleaseConfidence as d, evaluateReleaseConfidence as f, Verdict as g, JudgeReplayGateArgs as h, ReleaseConfidenceInput as i, wilcoxonSignedRank as k, ReleaseConfidenceThresholds as l, BootstrapResult as m, ReleaseConfidenceAxis as n, ReleaseConfidenceMetrics as o, BootstrapOptions as p, ReleaseConfidenceAxisName as r, ReleaseConfidenceScorecard as s, ActionableSideInfo as t, ReleaseTraceEvidence as u, judgeReplayGate as v, RankTestMethodRequest as w, MANN_WHITNEY_EXACT_MAX_WORK as x, DEFAULT_PERMUTATIONS as y };
|
|
313
|
-
//# sourceMappingURL=release-confidence-
|
|
313
|
+
//# sourceMappingURL=release-confidence-BcqeQTHW.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"release-confidence-
|
|
1
|
+
{"version":3,"file":"release-confidence-BcqeQTHW.d.ts","names":[],"sources":["../src/statistics/rank-tests.ts","../src/promotion-gate.ts","../src/release-confidence.ts"],"mappings":";;;;;;KA4BY;;;;;KAMA;UAEK;;EAEf,SAAS;;EAET;;;EAGA;;;cAIW;;cAEA;;cAEA;;cAEA;UAEI;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER;;;;;;;;;;;;;;iBAec,aACd,aACA,aACA,OAAM,kBACL;UAuFc;;;EAGf;;EAEA;;EAEA,QAAQ;;EAER;;EAEA;;;;;;;;;;;;;;;iBAgBc,mBACd,kBACA,iBACA,OAAM,kBACL;;;;;;;;;;;;;;;;;;;;;;;;;;;KC/KS;UAEK;EACf;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA,SAAS;;UAGM;;EAEf;;EAEA;;;;;;EAMA;;;EAGA;;;;;;;;;iBAUc,YACd,oBACA,qBACA,UAAS,mBACR;;;;;;;;;;;;;;;;UA0Gc,oBAAoB;EACnC,iBAAiB;EACjB,kBAAkB;;EAElB,QAAQ,QAAQ,YAAY;EAC5B;EACA;;EAEA;;EAEA;;;;;iBAMoB,gBAAgB,SACpC,MAAM,oBAAoB,WACzB,QAAQ;EAAoB;EAAyB;;;;;KC7K5C;;UAGK;;EAEf;;EAEA;EACA,WAAW;;EAEX;;EAEA;;EAEA;;EAEA;EACA,WAAW;;KAGD;KACA;UAQK;EACf;EACA;EACA,QAAQ;EACR;EACA;EACA;EACA;EACA;;EAEA,eAAe;EACf,MAAM;EACN,WAAW;;UAGI;;EAEf;EACA;EACA;EACA;;EAEA;EACA;EACA;;EAEA;EACA;EACA;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA;EACA,UAAU;EACV,qBAAqB;EACrB,gBAAgB;EAChB,kBAAkB;EAClB,eAAe;EACf,aAAa;;UAGE;EACf,MAAM;EACN,QAAQ;EACR;EACA;;UAGe;EACf,MAAM;EACN;EACA;EACA;;UAGe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,aAAa,OAAO;EACpB,cAAc;EACd,oBAAoB,QAAQ,OAAO;EACnC,0BAA0B;;;;;;;EAO1B;;UAGe;EACf;EACA;EACA;EACA,QAAQ;EACR;EACA,MAAM;EACN,QAAQ;EACR,SAAS;EACT,SAAS;EACT,cAAc;EACd;;iBAkBc,0BACd,OAAO,yBACN;iBA4Ha,wBAAwB,OAAO,yBAAyB"}
|
|
@@ -2,7 +2,7 @@ import { c as VerificationError, s as ValidationError } from "./errors-Dngq5h35.
|
|
|
2
2
|
import { t as mulberry32 } from "./random-Dn5fPWkt.js";
|
|
3
3
|
import { t as FAILURE_CLASSES } from "./schema-CSf6qWgZ.js";
|
|
4
4
|
import { r as observedSplitScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
|
|
5
|
-
import { c as validateRunRecord } from "./run-record-
|
|
5
|
+
import { c as validateRunRecord } from "./run-record-DualPTn2.js";
|
|
6
6
|
//#region src/promotion-gate.ts
|
|
7
7
|
/**
|
|
8
8
|
* Bootstrap-CI promotion gate.
|
|
@@ -523,4 +523,4 @@ function fmt(x) {
|
|
|
523
523
|
//#endregion
|
|
524
524
|
export { judgeReplayGate as i, evaluateReleaseConfidence as n, bootstrapCi as r, assertReleaseConfidence as t };
|
|
525
525
|
|
|
526
|
-
//# sourceMappingURL=release-confidence-
|
|
526
|
+
//# sourceMappingURL=release-confidence-DMg8n18l.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"release-confidence-BcGCclTB.js","names":[],"sources":["../src/promotion-gate.ts","../src/release-confidence.ts"],"sourcesContent":["/**\n * Bootstrap-CI promotion gate.\n *\n * In any iterative-improvement loop (GEPA, prompt evolution, dataset\n * curation), the question is \"did this generation actually improve, or are\n * we celebrating noise?\". With small N and noisy outcomes, point-estimate\n * deltas lie. Bootstrap confidence intervals tell the operator whether the\n * delta is real before code or prompts get promoted.\n *\n * This module is pure functions — no I/O, no model calls. Easy to unit-test\n * and to compose into any verdict gate.\n *\n * Default gate:\n * - Bootstrap mean baseline vs candidate (1k resamples).\n * - Compute the delta distribution; pass if the lower CI bound > 0.\n * - Tunable confidence (default 95%) and resample count.\n *\n * Verdict semantics intentionally match the existing `experiments.jsonl`\n * vocabulary:\n * - ADVANCE: candidate's CI lower bound > baseline mean (real win)\n * - KEEP: overlap, but candidate point estimate >= baseline (neutral)\n * - REVERT: candidate's CI upper bound < baseline mean (real regression)\n * - INCONCLUSIVE: not enough samples or CI straddles zero with no signal\n */\n\nimport { mulberry32 } from './statistics'\n\nexport type Verdict = 'ADVANCE' | 'KEEP' | 'REVERT' | 'INCONCLUSIVE'\n\nexport interface BootstrapResult {\n baselineMean: number\n candidateMean: number\n /** candidateMean - baselineMean, point estimate. */\n delta: number\n /** Lower bound of the (1 - alpha) CI on the delta. */\n ciLower: number\n /** Upper bound of the (1 - alpha) CI on the delta. */\n ciUpper: number\n /** Number of bootstrap resamples used. */\n iterations: number\n alpha: number\n verdict: Verdict\n}\n\nexport interface BootstrapOptions {\n /** Confidence level alpha (default 0.05 → 95% CI). */\n alpha?: number\n /** Number of resamples (default 1000). */\n iterations?: number\n /**\n * Minimum total samples (baseline + candidate) below which we always\n * return INCONCLUSIVE — bootstrap with too few samples is meaningless.\n * Default 6 (combined).\n */\n minTotalSamples?: number\n /** RNG seed for reproducibility. Absent, the seed is derived from the\n * compared arms, so the same inputs reproduce the same verdict. */\n seed?: number\n}\n\n/**\n * Compute the bootstrap CI on (candidateMean - baselineMean) and a verdict.\n *\n * Uses simple percentile bootstrap on the difference of resampled means.\n * That's the standard non-parametric primitive — no distributional\n * assumptions, robust to skew, easy to reason about.\n */\nexport function bootstrapCi(\n baseline: number[],\n candidate: number[],\n options: BootstrapOptions = {},\n): BootstrapResult {\n const alpha = options.alpha ?? 0.05\n const iterations = options.iterations ?? 1000\n const minTotal = options.minTotalSamples ?? 6\n const rng = mulberry32(options.seed ?? hashSeed(baseline, candidate))\n\n const baselineMean = mean(baseline)\n const candidateMean = mean(candidate)\n const delta = candidateMean - baselineMean\n\n if (\n baseline.length + candidate.length < minTotal ||\n baseline.length === 0 ||\n candidate.length === 0\n ) {\n return {\n baselineMean,\n candidateMean,\n delta,\n ciLower: -Infinity,\n ciUpper: Infinity,\n iterations: 0,\n alpha,\n verdict: 'INCONCLUSIVE',\n }\n }\n\n const deltas: number[] = new Array(iterations)\n for (let i = 0; i < iterations; i++) {\n const bResample = resample(baseline, rng)\n const cResample = resample(candidate, rng)\n deltas[i] = mean(cResample) - mean(bResample)\n }\n deltas.sort((a, b) => a - b)\n const lowerIdx = Math.floor((alpha / 2) * iterations)\n const upperIdx = Math.floor((1 - alpha / 2) * iterations) - 1\n const ciLower = deltas[Math.max(0, lowerIdx)]!\n const ciUpper = deltas[Math.min(iterations - 1, upperIdx)]!\n\n let verdict: Verdict\n // A ZERO-WIDTH interval is an absence of evidence, not a certainty. Constant\n // arms make every resample identical, so pass/fail data promotes on nothing:\n // [0,0,0] vs [1,1,1] is six samples, clears the `minTotalSamples` floor, and\n // yields [1, 1] — an \"ADVANCE\" carrying no information about how far the\n // estimate could be wrong. Same rule as `pairedDeltaTest`, one estimator over.\n if (!Number.isFinite(ciLower) || !Number.isFinite(ciUpper) || ciLower === ciUpper) {\n verdict = 'INCONCLUSIVE'\n } else if (ciLower > 0) verdict = 'ADVANCE'\n else if (ciUpper < 0) verdict = 'REVERT'\n else if (delta >= 0) verdict = 'KEEP'\n else verdict = 'INCONCLUSIVE'\n\n return {\n baselineMean,\n candidateMean,\n delta,\n ciLower,\n ciUpper,\n iterations,\n alpha,\n verdict,\n }\n}\n\nfunction mean(xs: number[]): number {\n if (xs.length === 0) return 0\n let s = 0\n for (const x of xs) s += x\n return s / xs.length\n}\n\nfunction resample(xs: number[], rng: () => number): number[] {\n const out = new Array(xs.length)\n for (let i = 0; i < xs.length; i++) out[i] = xs[Math.floor(rng() * xs.length)]\n return out\n}\n\n/** Stable seed derived from the inputs — same data → same CI bounds. */\nfunction hashSeed(a: number[], b: number[]): number {\n let h = 2166136261\n for (const x of [...a, ...b]) {\n const view = new Float64Array([x])\n const bytes = new Uint8Array(view.buffer)\n for (const byte of bytes) {\n h ^= byte\n h = Math.imul(h, 16777619)\n }\n }\n return h >>> 0\n}\n\n/**\n * Judge-replay promotion gate.\n *\n * The cheap inner-loop judge that drives an evolution run is by definition\n * fast and noisy. When you're about to promote a winning variant to the\n * canonical default, you want a STRONGER judge (a more expensive model, a\n * human grader, a separately-trained reward model) to confirm the win\n * generalises beyond the inner loop.\n *\n * This helper takes raw winner + baseline outputs, scores both through the\n * stronger judge, and applies `bootstrapCi`. ADVANCE means the stronger\n * judge agrees the winner is real with the configured confidence. Doesn't\n * matter what shape your \"output\" is — pass a string, an object, anything\n * the judge can read.\n */\nexport interface JudgeReplayGateArgs<TOutput> {\n baselineOutputs: TOutput[]\n candidateOutputs: TOutput[]\n /** Stronger judge — async to allow LLM calls. Return a 0..N scalar score. */\n judge: (output: TOutput) => Promise<number> | number\n alpha?: number\n iterations?: number\n /** RNG seed for reproducibility. */\n seed?: number\n /** Maximum concurrent judge calls. Default 4. */\n judgeConcurrency?: number\n}\n\n/**\n * Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.\n */\nexport async function judgeReplayGate<TOutput>(\n args: JudgeReplayGateArgs<TOutput>,\n): Promise<BootstrapResult & { baselineSamples: number; candidateSamples: number }> {\n const concurrency = args.judgeConcurrency ?? 4\n const baselineScores = await scoreAll(args.baselineOutputs, args.judge, concurrency)\n const candidateScores = await scoreAll(args.candidateOutputs, args.judge, concurrency)\n const ci = bootstrapCi(baselineScores, candidateScores, {\n ...(args.alpha !== undefined ? { alpha: args.alpha } : {}),\n ...(args.iterations !== undefined ? { iterations: args.iterations } : {}),\n ...(args.seed !== undefined ? { seed: args.seed } : {}),\n })\n return {\n ...ci,\n baselineSamples: baselineScores.length,\n candidateSamples: candidateScores.length,\n }\n}\n\nasync function scoreAll<TOutput>(\n outputs: TOutput[],\n judge: (output: TOutput) => Promise<number> | number,\n concurrency: number,\n): Promise<number[]> {\n const results: number[] = new Array(outputs.length)\n let next = 0\n async function worker(): Promise<void> {\n while (true) {\n const i = next++\n if (i >= outputs.length) return\n const v = await judge(outputs[i]!)\n results[i] = Number.isFinite(v) ? v : 0\n }\n }\n await Promise.all(Array.from({ length: Math.max(1, concurrency) }, () => worker()))\n return results\n}\n","/**\n * Release confidence gate.\n *\n * This is the production-facing composition layer over the lower-level\n * primitives:\n * - Dataset manifests prove corpus/version coverage.\n * - RunRecord rows prove reproducible search/holdout outcomes.\n * - Multi-shot trace evidence carries turn counts and ASI diagnostics.\n * - HeldOutGate decisions remain the paired promotion authority.\n *\n * The gate is intentionally pure and conservative. Missing declared evidence\n * fails closed instead of being treated as a neutral zero.\n */\n\nimport type { DatasetManifest, DatasetScenario, DatasetSplit } from './dataset'\nimport { ValidationError, VerificationError } from './errors'\nimport type { GateDecision } from './held-out-gate'\nimport { isRealnessGated, observedSplitScore } from './rollout/reward'\nimport { type RunRecord, type RunSplitTag, validateRunRecord } from './run-record'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Severity of an actionable finding attached to a run/trace. */\nexport type AsiSeverity = 'info' | 'warning' | 'error' | 'critical'\n\n/** Actionable side-info — a diagnosed finding the loop can act on. */\nexport interface ActionableSideInfo {\n /** Stable expectation/check id when available. */\n expectationId?: string\n /** Human-readable diagnosis of what happened. */\n message: string\n severity?: AsiSeverity\n /** Concrete trace excerpt, file path, tool call, screenshot id, etc. */\n evidence?: string\n /** Prompt/tool/context surface likely responsible. */\n responsibleSurface?: string\n /** Suggested fix in natural language. */\n suggestion?: string\n /** Whether this expectation was satisfied. Defaults to false for ASI rows. */\n matched?: boolean\n metadata?: Record<string, unknown>\n}\n\nexport type ReleaseConfidenceStatus = 'pass' | 'warn' | 'fail'\nexport type ReleaseConfidenceAxisName =\n | 'corpus'\n | 'quality'\n | 'reliability'\n | 'generalization'\n | 'diagnostics'\n | 'efficiency'\n\nexport interface ReleaseTraceEvidence {\n scenarioId: string\n candidateId?: string\n split?: RunSplitTag\n score?: number\n ok?: boolean\n turnCount?: number\n costUsd?: number\n durationMs?: number\n /** Canonical task-failure class. Free-form detail belongs in ASI. */\n failureClass?: FailureClass\n asi?: ActionableSideInfo[]\n metadata?: Record<string, unknown>\n}\n\nexport interface ReleaseConfidenceThresholds {\n /** Require a Dataset manifest or explicit scenarios. Default true. */\n requireCorpus?: boolean\n minScenarioCount?: number\n minSearchRuns?: number\n minHoldoutRuns?: number\n /** Require at least one holdout scenario/run. Default true. */\n requireHoldout?: boolean\n minPassRate?: number\n minMeanScore?: number\n /** Search mean may exceed holdout mean by at most this much. */\n maxOverfitGap?: number\n maxMeanCostUsd?: number\n maxP95WallMs?: number\n /** Low-score/failed rows must carry ASI. Default true. */\n requireAsiForFailures?: boolean\n /** Score below this is considered a failure for ASI coverage. Default 0.5. */\n failureScoreThreshold?: number\n}\n\nexport interface ReleaseConfidenceInput {\n target: string\n candidateId?: string\n baselineId?: string\n dataset?: DatasetManifest\n scenarios?: readonly DatasetScenario[]\n runs?: readonly RunRecord[]\n traces?: readonly ReleaseTraceEvidence[]\n gateDecision?: GateDecision | null\n thresholds?: ReleaseConfidenceThresholds\n}\n\nexport interface ReleaseConfidenceAxis {\n name: ReleaseConfidenceAxisName\n status: ReleaseConfidenceStatus\n score: number | null\n detail: string\n}\n\nexport interface ReleaseConfidenceIssue {\n axis: ReleaseConfidenceAxisName\n severity: 'critical' | 'warning'\n code: string\n detail: string\n}\n\nexport interface ReleaseConfidenceMetrics {\n scenarioCount: number\n /** Search rows with a finite search score. */\n searchRuns: number\n /** Holdout rows with a finite holdout score. */\n holdoutRuns: number\n /** Runs with neither a split-matched score nor an explicit task failure. */\n unscoredRuns: number\n /** Run rows, or trace rows when no runs exist, with no classified terminal result. */\n unclassifiedTerminalRuns: number\n /** Run rows, or trace rows when no runs exist, that ended unsuccessfully. */\n terminalFailureRuns: number\n /** Success fraction when every run or fallback trace row has a classified result. */\n reliabilityRate: number | null\n passRate: number | null\n meanScore: number | null\n searchMeanScore: number | null\n holdoutMeanScore: number | null\n overfitGap: number | null\n meanCostUsd: number | null\n p95WallMs: number | null\n failedRows: number\n failuresWithAsi: number\n singleShotTraces: number\n multiShotTraces: number\n splitCounts: Record<DatasetSplit, number>\n domainCounts: Record<string, number>\n failureClassCounts: Partial<Record<FailureClass, number>>\n responsibleSurfaceCounts: Record<string, number>\n /**\n * Runs excluded from `passRate` because the authenticity gate flagged them as\n * gamed. Surfaced, never silent: a release whose pass rate is computed over a\n * shrunken denominator has to say by how much, or the exclusion is just a\n * different way of hiding the same runs.\n */\n realnessGatedRuns: number\n}\n\nexport interface ReleaseConfidenceScorecard {\n target: string\n candidateId: string | null\n baselineId: string | null\n status: ReleaseConfidenceStatus\n promote: boolean\n axes: ReleaseConfidenceAxis[]\n issues: ReleaseConfidenceIssue[]\n metrics: ReleaseConfidenceMetrics\n dataset: DatasetManifest | null\n gateDecision: GateDecision | null\n summary: string\n}\n\nconst DEFAULT_THRESHOLDS: Required<ReleaseConfidenceThresholds> = {\n requireCorpus: true,\n minScenarioCount: 1,\n minSearchRuns: 1,\n minHoldoutRuns: 1,\n requireHoldout: true,\n minPassRate: 0.8,\n minMeanScore: 0.7,\n maxOverfitGap: 0.15,\n maxMeanCostUsd: Number.POSITIVE_INFINITY,\n maxP95WallMs: Number.POSITIVE_INFINITY,\n requireAsiForFailures: true,\n failureScoreThreshold: 0.5,\n}\n\nexport function evaluateReleaseConfidence(\n input: ReleaseConfidenceInput,\n): ReleaseConfidenceScorecard {\n const thresholds = { ...DEFAULT_THRESHOLDS, ...input.thresholds }\n const candidateId = input.candidateId ?? null\n const runs = filterCandidate(\n (input.runs ?? []).map(validateRunRecord),\n candidateId,\n input.baselineId,\n )\n const traces = filterTraceCandidate(\n (input.traces ?? []).map(validateReleaseTraceEvidence),\n candidateId,\n input.baselineId,\n )\n const scenarios = input.scenarios ?? []\n const scenarioCount = input.dataset?.scenarioCount ?? scenarios.length\n const splitCounts = input.dataset?.splitCounts ?? countScenarioSplits(scenarios)\n const searchScores = scoresFor(runs, 'search')\n const holdoutScores = scoresFor(runs, 'holdout')\n const runScores = runs.map(runSplitScore).filter(isFiniteNumber)\n const traceScores = traces.map((t) => t.score).filter(isFiniteNumber)\n const scoreUniverse = runs.length > 0 ? runScores : traceScores\n const qualityRuns = runs.filter((run) => runSplitScore(run) !== undefined)\n const unscoredRuns = runs.filter(\n (run) =>\n runSplitScore(run) === undefined &&\n !hasExplicitTaskFailure(run) &&\n !isFailedTerminalOutcome(run.terminalOutcome),\n ).length\n // Realness-gated runs are EXCLUDED from the pass rate — numerator AND\n // denominator — and the count of what was dropped ships beside the rate as\n // `metrics.realnessGatedRuns`. Counting a gamed run as a pass made faking a\n // success the cheapest way to improve a release scorecard; scoring it 0\n // instead would be the other error, silently deflating the rate with a run\n // the gate says carries no usable verdict at all.\n const honestRuns = runs.filter((run) => !isRealnessGated(run))\n const passOutcomes =\n runs.length > 0\n ? honestRuns.map((run) => runPassOutcome(run, thresholds.failureScoreThreshold))\n : traces.map((trace) => tracePassOutcome(trace, thresholds.failureScoreThreshold))\n const reliabilityRows =\n runs.length > 0\n ? runs.map((run) => terminalSuccess(run.terminalOutcome))\n : traces.map((trace) => trace.ok)\n const unclassifiedTerminalRuns = reliabilityRows.filter((outcome) => outcome === undefined).length\n const terminalFailureRuns = reliabilityRows.filter((outcome) => outcome === false).length\n const reliabilityRate =\n reliabilityRows.length === 0 || unclassifiedTerminalRuns > 0\n ? null\n : (reliabilityRows.length - terminalFailureRuns) / reliabilityRows.length\n const searchRuns = qualityRuns.filter((r) => r.splitTag === 'search').length\n const holdoutRuns = qualityRuns.filter((r) => r.splitTag === 'holdout').length\n const failed = failedRows(runs, traces, thresholds.failureScoreThreshold)\n const searchMeanScore = meanOrNull(searchScores)\n const holdoutMeanScore = meanOrNull(holdoutScores)\n const runCosts = runs.flatMap((run) =>\n run.costProvenance.kind === 'uncaptured' ? [] : [run.costProvenance.usd],\n )\n const traceCosts = traces.map((trace) => trace.costUsd).filter(isFiniteNumber)\n const meanCostUsd =\n runs.length > 0\n ? runCosts.length === runs.length\n ? meanOrNull(runCosts)\n : null\n : meanOrNull(traceCosts)\n const wallTimes =\n runs.length > 0\n ? runs.map((run) => run.wallMs)\n : traces.map((trace) => trace.durationMs).filter(isFiniteNumber)\n const metrics: ReleaseConfidenceMetrics = {\n scenarioCount,\n searchRuns,\n holdoutRuns,\n unscoredRuns,\n unclassifiedTerminalRuns,\n terminalFailureRuns,\n reliabilityRate,\n passRate: passOutcomeRate(passOutcomes),\n realnessGatedRuns: runs.length - honestRuns.length,\n meanScore: meanOrNull(scoreUniverse),\n searchMeanScore,\n holdoutMeanScore,\n overfitGap: diffOrNull(searchMeanScore, holdoutMeanScore),\n meanCostUsd,\n p95WallMs: percentileOrNull(wallTimes, 0.95),\n failedRows: failed.length,\n failuresWithAsi: failed.filter((row) => row.hasAsi).length,\n singleShotTraces: traces.filter((t) => t.turnCount === 1).length,\n multiShotTraces: traces.filter((t) => (t.turnCount ?? 0) > 1).length,\n splitCounts,\n domainCounts: countDomains(scenarios),\n failureClassCounts: countFailureClasses(runs, traces, thresholds.failureScoreThreshold),\n responsibleSurfaceCounts: countResponsibleSurfaces(traces),\n }\n\n const issues: ReleaseConfidenceIssue[] = []\n checkCorpus(input, thresholds, metrics, issues)\n checkQuality(thresholds, metrics, issues)\n checkReliability(metrics, issues)\n checkGeneralization(input.gateDecision ?? null, thresholds, metrics, issues)\n checkDiagnostics(thresholds, metrics, issues)\n checkEfficiency(thresholds, metrics, issues)\n\n const axes = buildAxes(metrics, thresholds, issues)\n const status = issues.some((i) => i.severity === 'critical')\n ? 'fail'\n : issues.length > 0\n ? 'warn'\n : 'pass'\n\n return {\n target: input.target,\n candidateId,\n baselineId: input.baselineId ?? null,\n status,\n promote: status === 'pass' && (input.gateDecision ? input.gateDecision.promote : true),\n axes,\n issues,\n metrics,\n dataset: input.dataset ?? null,\n gateDecision: input.gateDecision ?? null,\n summary: renderSummary(input.target, status, metrics, issues),\n }\n}\n\nexport function assertReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard {\n const scorecard = evaluateReleaseConfidence(input)\n if (scorecard.status === 'fail') {\n throw new VerificationError(scorecard.summary)\n }\n return scorecard\n}\n\nfunction filterCandidate(\n runs: readonly RunRecord[],\n candidateId: string | null,\n baselineId?: string,\n): RunRecord[] {\n if (candidateId) return runs.filter((r) => r.candidateId === candidateId)\n if (baselineId) return runs.filter((r) => r.candidateId !== baselineId)\n return [...runs]\n}\n\nfunction filterTraceCandidate(\n traces: readonly ReleaseTraceEvidence[],\n candidateId: string | null,\n baselineId?: string,\n): ReleaseTraceEvidence[] {\n if (candidateId)\n return traces.filter((t) => t.candidateId === undefined || t.candidateId === candidateId)\n if (baselineId)\n return traces.filter((t) => t.candidateId === undefined || t.candidateId !== baselineId)\n return [...traces]\n}\n\nfunction validateReleaseTraceEvidence(\n trace: ReleaseTraceEvidence,\n index: number,\n): ReleaseTraceEvidence {\n const value = trace as unknown as Record<string, unknown>\n if (Object.hasOwn(value, 'failureMode')) {\n throw new ValidationError(\n `traces[${index}].failureMode is not supported; use canonical failureClass`,\n )\n }\n if (trace.failureClass !== undefined && !FAILURE_CLASSES.includes(trace.failureClass)) {\n throw new ValidationError(\n `traces[${index}].failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n )\n }\n return trace\n}\n\nfunction checkCorpus(\n input: ReleaseConfidenceInput,\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (thresholds.requireCorpus && !input.dataset && (input.scenarios?.length ?? 0) === 0) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'missing_corpus',\n detail: 'No Dataset manifest or scenarios supplied.',\n })\n }\n if (metrics.scenarioCount < thresholds.minScenarioCount) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'few_scenarios',\n detail: `${metrics.scenarioCount} scenario(s) < min ${thresholds.minScenarioCount}.`,\n })\n }\n if (thresholds.requireHoldout && metrics.splitCounts.holdout === 0) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'missing_holdout_split',\n detail: 'Corpus has no holdout scenarios.',\n })\n }\n}\n\nfunction checkQuality(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (metrics.searchRuns < thresholds.minSearchRuns) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'few_search_runs',\n detail: `${metrics.searchRuns} search run(s) < min ${thresholds.minSearchRuns}.`,\n })\n }\n if (metrics.unscoredRuns > 0) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'unscored_runs',\n detail: `${metrics.unscoredRuns} supplied run(s) have no task result.`,\n })\n }\n if (metrics.passRate === null || metrics.meanScore === null) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'missing_quality_scores',\n detail: 'No task-quality scores are available for pass-rate and mean-score checks.',\n })\n }\n if (metrics.passRate !== null && metrics.passRate < thresholds.minPassRate) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'low_pass_rate',\n detail: `passRate ${fmt(metrics.passRate)} < ${fmt(thresholds.minPassRate)}.`,\n })\n }\n if (metrics.meanScore !== null && metrics.meanScore < thresholds.minMeanScore) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'low_mean_score',\n detail: `meanScore ${fmt(metrics.meanScore)} < ${fmt(thresholds.minMeanScore)}.`,\n })\n }\n}\n\nfunction checkReliability(\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (metrics.reliabilityRate === null) {\n issues.push({\n axis: 'reliability',\n severity: 'critical',\n code: 'missing_reliability_evidence',\n detail:\n metrics.unclassifiedTerminalRuns > 0\n ? `${metrics.unclassifiedTerminalRuns} supplied run(s) have no classified terminal result.`\n : 'No classified terminal results are available.',\n })\n }\n if (metrics.terminalFailureRuns > 0) {\n issues.push({\n axis: 'reliability',\n severity: 'critical',\n code: 'terminal_run_failures',\n detail: `${metrics.terminalFailureRuns} run(s) ended failed, cancelled, or incomplete.`,\n })\n }\n}\n\nfunction checkGeneralization(\n gateDecision: GateDecision | null,\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (thresholds.requireHoldout && metrics.holdoutRuns < thresholds.minHoldoutRuns) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: 'few_holdout_runs',\n detail: `${metrics.holdoutRuns} holdout run(s) < min ${thresholds.minHoldoutRuns}.`,\n })\n }\n if (metrics.overfitGap !== null && metrics.overfitGap > thresholds.maxOverfitGap) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: 'overfit_gap',\n detail: `search-holdout gap ${fmt(metrics.overfitGap)} > ${fmt(thresholds.maxOverfitGap)}.`,\n })\n }\n if (gateDecision && !gateDecision.promote) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: `gate_${gateDecision.rejectionCode ?? 'reject'}`,\n detail: gateDecision.reason,\n })\n }\n}\n\nfunction checkDiagnostics(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (!thresholds.requireAsiForFailures) return\n if (metrics.failedRows > metrics.failuresWithAsi) {\n issues.push({\n axis: 'diagnostics',\n severity: 'critical',\n code: 'missing_failure_asi',\n detail: `${metrics.failedRows - metrics.failuresWithAsi} failed row(s) have no actionable side information.`,\n })\n }\n}\n\nfunction checkEfficiency(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (Number.isFinite(thresholds.maxMeanCostUsd) && metrics.meanCostUsd === null) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'missing_cost',\n detail: 'A finite cost limit was configured but no cost evidence is available.',\n })\n } else if (metrics.meanCostUsd !== null && metrics.meanCostUsd > thresholds.maxMeanCostUsd) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'cost_budget',\n detail: `meanCostUsd ${fmt(metrics.meanCostUsd)} > ${fmt(thresholds.maxMeanCostUsd)}.`,\n })\n }\n if (Number.isFinite(thresholds.maxP95WallMs) && metrics.p95WallMs === null) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'missing_latency',\n detail: 'A finite latency limit was configured but no latency evidence is available.',\n })\n } else if (metrics.p95WallMs !== null && metrics.p95WallMs > thresholds.maxP95WallMs) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'latency_budget',\n detail: `p95WallMs ${fmt(metrics.p95WallMs)} > ${fmt(thresholds.maxP95WallMs)}.`,\n })\n }\n}\n\nfunction buildAxes(\n metrics: ReleaseConfidenceMetrics,\n thresholds: Required<ReleaseConfidenceThresholds>,\n issues: ReleaseConfidenceIssue[],\n): ReleaseConfidenceAxis[] {\n return [\n axis(\n 'corpus',\n issues,\n bounded(metrics.scenarioCount / Math.max(1, thresholds.minScenarioCount)),\n `${metrics.scenarioCount} scenarios; holdout=${metrics.splitCounts.holdout}`,\n ),\n axis(\n 'quality',\n issues,\n metrics.passRate === null || metrics.meanScore === null\n ? null\n : Math.min(metrics.passRate, metrics.meanScore),\n `passRate=${fmt(metrics.passRate)} meanScore=${fmt(metrics.meanScore)}`,\n ),\n axis(\n 'reliability',\n issues,\n metrics.reliabilityRate,\n `successRate=${fmt(metrics.reliabilityRate)} terminalFailures=${metrics.terminalFailureRuns} unclassified=${metrics.unclassifiedTerminalRuns}`,\n ),\n axis(\n 'generalization',\n issues,\n gapScore(metrics.overfitGap, thresholds.maxOverfitGap),\n `holdoutRuns=${metrics.holdoutRuns} overfitGap=${fmt(metrics.overfitGap)}`,\n ),\n axis(\n 'diagnostics',\n issues,\n metrics.failedRows === 0 ? 1 : metrics.failuresWithAsi / metrics.failedRows,\n `failuresWithAsi=${metrics.failuresWithAsi}/${metrics.failedRows}`,\n ),\n axis(\n 'efficiency',\n issues,\n efficiencyScore(metrics, thresholds),\n `meanCostUsd=${fmt(metrics.meanCostUsd)} p95WallMs=${fmt(metrics.p95WallMs)}`,\n ),\n ]\n}\n\nfunction axis(\n name: ReleaseConfidenceAxisName,\n issues: ReleaseConfidenceIssue[],\n score: number | null,\n detail: string,\n): ReleaseConfidenceAxis {\n const own = issues.filter((i) => i.axis === name)\n const status = own.some((i) => i.severity === 'critical')\n ? 'fail'\n : own.length > 0\n ? 'warn'\n : 'pass'\n return { name, status, score: score === null ? null : bounded(score), detail }\n}\n\nfunction countScenarioSplits(scenarios: readonly DatasetScenario[]): Record<DatasetSplit, number> {\n const counts: Record<DatasetSplit, number> = { train: 0, dev: 0, test: 0, holdout: 0 }\n for (const scenario of scenarios) counts[scenario.split ?? 'train']++\n return counts\n}\n\nfunction countDomains(scenarios: readonly DatasetScenario[]): Record<string, number> {\n const out: Record<string, number> = {}\n for (const scenario of scenarios) {\n const domain = scenario.tags?.domain ?? scenario.tags?.category ?? 'uncategorized'\n out[domain] = (out[domain] ?? 0) + 1\n }\n return out\n}\n\nfunction countFailureClasses(\n runs: readonly RunRecord[],\n traces: readonly ReleaseTraceEvidence[],\n threshold: number,\n): Partial<Record<FailureClass, number>> {\n const out: Partial<Record<FailureClass, number>> = {}\n for (const run of runs) {\n // Ungated: a failure-mode census counts what the runs REPORTED, which is\n // the only way a gamed run's inflated score is visible at all. It is not a\n // promotion number — `passRate` is, and that one excludes gated runs and\n // publishes the excluded count as `metrics.realnessGatedRuns`.\n if (runPassOutcome(run, threshold) === false) {\n const failureClass =\n run.failureClass !== undefined && run.failureClass !== 'success'\n ? run.failureClass\n : 'unknown'\n out[failureClass] = (out[failureClass] ?? 0) + 1\n }\n }\n for (const trace of traces) {\n if (tracePassOutcome(trace, threshold) === false) {\n const failureClass =\n trace.failureClass !== undefined && trace.failureClass !== 'success'\n ? trace.failureClass\n : 'unknown'\n out[failureClass] = (out[failureClass] ?? 0) + 1\n }\n }\n return out\n}\n\nfunction countResponsibleSurfaces(traces: readonly ReleaseTraceEvidence[]): Record<string, number> {\n const out: Record<string, number> = {}\n for (const trace of traces) {\n for (const asi of trace.asi ?? []) {\n const surface = asi.responsibleSurface ?? 'unknown'\n out[surface] = (out[surface] ?? 0) + 1\n }\n }\n return out\n}\n\nfunction failedRows(\n runs: readonly RunRecord[],\n traces: readonly ReleaseTraceEvidence[],\n threshold: number,\n): Array<{ hasAsi: boolean }> {\n const out: Array<{ hasAsi: boolean }> = []\n for (const run of runs) {\n if (runPassOutcome(run, threshold) === false) {\n const asiMetric = run.outcome.raw.asi\n out.push({ hasAsi: typeof asiMetric === 'number' && asiMetric > 0 })\n }\n }\n for (const trace of traces) {\n if (tracePassOutcome(trace, threshold) === false) {\n out.push({ hasAsi: (trace.asi?.length ?? 0) > 0 })\n }\n }\n return out\n}\n\nfunction passOutcomeRate(outcomes: readonly (boolean | null)[]): number | null {\n const classified = outcomes.filter((outcome): outcome is boolean => outcome !== null)\n if (classified.length === 0) return null\n return classified.filter(Boolean).length / classified.length\n}\n\nfunction runPassOutcome(run: RunRecord, threshold: number): boolean | null {\n if (hasExplicitTaskFailure(run)) return false\n const score = runSplitScore(run)\n return score === undefined ? null : score >= threshold\n}\n\nfunction hasExplicitTaskFailure(run: RunRecord): boolean {\n return run.failureClass !== undefined && run.failureClass !== 'success'\n}\n\nfunction tracePassOutcome(trace: ReleaseTraceEvidence, threshold: number): boolean | null {\n if (trace.failureClass !== undefined && trace.failureClass !== 'success') return false\n if (trace.ok === false) return false\n if (isFiniteNumber(trace.score)) return trace.score >= threshold\n return trace.ok === true ? true : null\n}\n\nfunction isFailedTerminalOutcome(\n outcome: RunRecord['terminalOutcome'],\n): outcome is 'failed' | 'cancelled' | 'incomplete' {\n return outcome === 'failed' || outcome === 'cancelled' || outcome === 'incomplete'\n}\n\nfunction terminalSuccess(outcome: RunRecord['terminalOutcome']): boolean | undefined {\n if (outcome === 'succeeded') return true\n if (isFailedTerminalOutcome(outcome)) return false\n return undefined\n}\n\nfunction scoresFor(runs: readonly RunRecord[], split: RunSplitTag): number[] {\n return runs\n .filter((run) => run.splitTag === split)\n .map(runSplitScore)\n .filter(isFiniteNumber)\n}\n\n/**\n * RAW and split-exact (`observedSplitScore`): this feeds the per-split means,\n * the overfit gap, and the pass threshold — descriptions of what the runs\n * reported. A gamed run inflating them is visible next to `realnessGatedRuns`,\n * and the promotion number (`passRate`) excludes gated runs entirely.\n */\nfunction runSplitScore(run: RunRecord): number | undefined {\n const score = observedSplitScore(run, run.splitTag === 'holdout' ? 'holdout' : 'search')\n return isFiniteNumber(score) ? score : undefined\n}\n\nfunction meanOrNull(xs: readonly number[]): number | null {\n if (xs.length === 0) return null\n return xs.reduce((sum, x) => sum + x, 0) / xs.length\n}\n\nfunction percentileOrNull(xs: readonly number[], p: number): number | null {\n if (xs.length === 0) return null\n const sorted = [...xs].sort((a, b) => a - b)\n return sorted[Math.min(sorted.length - 1, Math.max(0, Math.ceil(p * sorted.length) - 1))]!\n}\n\nfunction isFiniteNumber(value: unknown): value is number {\n return typeof value === 'number' && Number.isFinite(value)\n}\n\nfunction diffOrNull(a: number | null, b: number | null): number | null {\n if (a === null || b === null) return null\n return a - b\n}\n\nfunction gapScore(gap: number | null, maxGap: number): number | null {\n if (gap === null) return null\n if (maxGap <= 0) return gap <= 0 ? 1 : 0\n return bounded(1 - Math.max(0, gap) / maxGap)\n}\n\nfunction efficiencyScore(\n metrics: ReleaseConfidenceMetrics,\n thresholds: Required<ReleaseConfidenceThresholds>,\n): number | null {\n const cost = Number.isFinite(thresholds.maxMeanCostUsd)\n ? metrics.meanCostUsd === null\n ? null\n : bounded(thresholds.maxMeanCostUsd / Math.max(metrics.meanCostUsd, 1e-12))\n : 1\n const latency = Number.isFinite(thresholds.maxP95WallMs)\n ? metrics.p95WallMs === null\n ? null\n : bounded(thresholds.maxP95WallMs / Math.max(metrics.p95WallMs, 1e-12))\n : 1\n if (cost === null || latency === null) return null\n return Math.min(cost, latency)\n}\n\nfunction bounded(x: number): number {\n if (!Number.isFinite(x)) return 0\n return Math.max(0, Math.min(1, x))\n}\n\nfunction renderSummary(\n target: string,\n status: ReleaseConfidenceStatus,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): string {\n const prefix = `release confidence ${status}: ${target}`\n const metricText = `scenarios=${metrics.scenarioCount} searchRuns=${metrics.searchRuns} holdoutRuns=${metrics.holdoutRuns} passRate=${fmt(metrics.passRate)} meanScore=${fmt(metrics.meanScore)}`\n if (issues.length === 0) return `${prefix}; ${metricText}`\n return `${prefix}; ${metricText}; issues=${issues.map((i) => i.code).join(',')}`\n}\n\nfunction fmt(x: number | null): string {\n if (x === null) return 'n/a'\n return x.toFixed(4)\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmEA,SAAgB,YACd,UACA,WACA,UAA4B,CAAC,GACZ;CACjB,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,aAAa,QAAQ,cAAc;CACzC,MAAM,WAAW,QAAQ,mBAAmB;CAC5C,MAAM,MAAM,WAAW,QAAQ,QAAQ,SAAS,UAAU,SAAS,CAAC;CAEpE,MAAM,eAAe,KAAK,QAAQ;CAClC,MAAM,gBAAgB,KAAK,SAAS;CACpC,MAAM,QAAQ,gBAAgB;CAE9B,IACE,SAAS,SAAS,UAAU,SAAS,YACrC,SAAS,WAAW,KACpB,UAAU,WAAW,GAErB,OAAO;EACL;EACA;EACA;EACA,SAAS;EACT,SAAS;EACT,YAAY;EACZ;EACA,SAAS;CACX;CAGF,MAAM,SAAmB,IAAI,MAAM,UAAU;CAC7C,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,KAAK;EACnC,MAAM,YAAY,SAAS,UAAU,GAAG;EAExC,OAAO,KAAK,KADM,SAAS,WAAW,GACb,CAAC,IAAI,KAAK,SAAS;CAC9C;CACA,OAAO,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3B,MAAM,WAAW,KAAK,MAAO,QAAQ,IAAK,UAAU;CACpD,MAAM,WAAW,KAAK,OAAO,IAAI,QAAQ,KAAK,UAAU,IAAI;CAC5D,MAAM,UAAU,OAAO,KAAK,IAAI,GAAG,QAAQ;CAC3C,MAAM,UAAU,OAAO,KAAK,IAAI,aAAa,GAAG,QAAQ;CAExD,IAAI;CAMJ,IAAI,CAAC,OAAO,SAAS,OAAO,KAAK,CAAC,OAAO,SAAS,OAAO,KAAK,YAAY,SACxE,UAAU;MACL,IAAI,UAAU,GAAG,UAAU;MAC7B,IAAI,UAAU,GAAG,UAAU;MAC3B,IAAI,SAAS,GAAG,UAAU;MAC1B,UAAU;CAEf,OAAO;EACL;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF;AACF;AAEA,SAAS,KAAK,IAAsB;CAClC,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,IAAI,KAAK;CACzB,OAAO,IAAI,GAAG;AAChB;AAEA,SAAS,SAAS,IAAc,KAA6B;CAC3D,MAAM,MAAM,IAAI,MAAM,GAAG,MAAM;CAC/B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,QAAQ,KAAK,IAAI,KAAK,GAAG,KAAK,MAAM,IAAI,IAAI,GAAG,MAAM;CAC5E,OAAO;AACT;;AAGA,SAAS,SAAS,GAAa,GAAqB;CAClD,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,CAAC,GAAG,GAAG,GAAG,CAAC,GAAG;EAC5B,MAAM,OAAO,IAAI,aAAa,CAAC,CAAC,CAAC;EACjC,MAAM,QAAQ,IAAI,WAAW,KAAK,MAAM;EACxC,KAAK,MAAM,QAAQ,OAAO;GACxB,KAAK;GACL,IAAI,KAAK,KAAK,GAAG,QAAQ;EAC3B;CACF;CACA,OAAO,MAAM;AACf;;;;AAiCA,eAAsB,gBACpB,MACkF;CAClF,MAAM,cAAc,KAAK,oBAAoB;CAC7C,MAAM,iBAAiB,MAAM,SAAS,KAAK,iBAAiB,KAAK,OAAO,WAAW;CACnF,MAAM,kBAAkB,MAAM,SAAS,KAAK,kBAAkB,KAAK,OAAO,WAAW;CAMrF,OAAO;EACL,GANS,YAAY,gBAAgB,iBAAiB;GACtD,GAAI,KAAK,UAAU,KAAA,IAAY,EAAE,OAAO,KAAK,MAAM,IAAI,CAAC;GACxD,GAAI,KAAK,eAAe,KAAA,IAAY,EAAE,YAAY,KAAK,WAAW,IAAI,CAAC;GACvE,GAAI,KAAK,SAAS,KAAA,IAAY,EAAE,MAAM,KAAK,KAAK,IAAI,CAAC;EACvD,CAEM;EACJ,iBAAiB,eAAe;EAChC,kBAAkB,gBAAgB;CACpC;AACF;AAEA,eAAe,SACb,SACA,OACA,aACmB;CACnB,MAAM,UAAoB,IAAI,MAAM,QAAQ,MAAM;CAClD,IAAI,OAAO;CACX,eAAe,SAAwB;EACrC,OAAO,MAAM;GACX,MAAM,IAAI;GACV,IAAI,KAAK,QAAQ,QAAQ;GACzB,MAAM,IAAI,MAAM,MAAM,QAAQ,EAAG;GACjC,QAAQ,KAAK,OAAO,SAAS,CAAC,IAAI,IAAI;EACxC;CACF;CACA,MAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,GAAG,WAAW,EAAE,SAAS,OAAO,CAAC,CAAC;CAClF,OAAO;AACT;;;AChEA,MAAM,qBAA4D;CAChE,eAAe;CACf,kBAAkB;CAClB,eAAe;CACf,gBAAgB;CAChB,gBAAgB;CAChB,aAAa;CACb,cAAc;CACd,eAAe;CACf,gBAAgB,OAAO;CACvB,cAAc,OAAO;CACrB,uBAAuB;CACvB,uBAAuB;AACzB;AAEA,SAAgB,0BACd,OAC4B;CAC5B,MAAM,aAAa;EAAE,GAAG;EAAoB,GAAG,MAAM;CAAW;CAChE,MAAM,cAAc,MAAM,eAAe;CACzC,MAAM,OAAO,iBACV,MAAM,QAAQ,CAAC,EAAA,CAAG,IAAI,iBAAiB,GACxC,aACA,MAAM,UACR;CACA,MAAM,SAAS,sBACZ,MAAM,UAAU,CAAC,EAAA,CAAG,IAAI,4BAA4B,GACrD,aACA,MAAM,UACR;CACA,MAAM,YAAY,MAAM,aAAa,CAAC;CACtC,MAAM,gBAAgB,MAAM,SAAS,iBAAiB,UAAU;CAChE,MAAM,cAAc,MAAM,SAAS,eAAe,oBAAoB,SAAS;CAC/E,MAAM,eAAe,UAAU,MAAM,QAAQ;CAC7C,MAAM,gBAAgB,UAAU,MAAM,SAAS;CAC/C,MAAM,YAAY,KAAK,IAAI,aAAa,CAAC,CAAC,OAAO,cAAc;CAC/D,MAAM,cAAc,OAAO,KAAK,MAAM,EAAE,KAAK,CAAC,CAAC,OAAO,cAAc;CACpE,MAAM,gBAAgB,KAAK,SAAS,IAAI,YAAY;CACpD,MAAM,cAAc,KAAK,QAAQ,QAAQ,cAAc,GAAG,MAAM,KAAA,CAAS;CACzE,MAAM,eAAe,KAAK,QACvB,QACC,cAAc,GAAG,MAAM,KAAA,KACvB,CAAC,uBAAuB,GAAG,KAC3B,CAAC,wBAAwB,IAAI,eAAe,CAChD,CAAC,CAAC;CAOF,MAAM,aAAa,KAAK,QAAQ,QAAQ,CAAC,gBAAgB,GAAG,CAAC;CAC7D,MAAM,eACJ,KAAK,SAAS,IACV,WAAW,KAAK,QAAQ,eAAe,KAAK,WAAW,qBAAqB,CAAC,IAC7E,OAAO,KAAK,UAAU,iBAAiB,OAAO,WAAW,qBAAqB,CAAC;CACrF,MAAM,kBACJ,KAAK,SAAS,IACV,KAAK,KAAK,QAAQ,gBAAgB,IAAI,eAAe,CAAC,IACtD,OAAO,KAAK,UAAU,MAAM,EAAE;CACpC,MAAM,2BAA2B,gBAAgB,QAAQ,YAAY,YAAY,KAAA,CAAS,CAAC,CAAC;CAC5F,MAAM,sBAAsB,gBAAgB,QAAQ,YAAY,YAAY,KAAK,CAAC,CAAC;CACnF,MAAM,kBACJ,gBAAgB,WAAW,KAAK,2BAA2B,IACvD,QACC,gBAAgB,SAAS,uBAAuB,gBAAgB;CACvE,MAAM,aAAa,YAAY,QAAQ,MAAM,EAAE,aAAa,QAAQ,CAAC,CAAC;CACtE,MAAM,cAAc,YAAY,QAAQ,MAAM,EAAE,aAAa,SAAS,CAAC,CAAC;CACxE,MAAM,SAAS,WAAW,MAAM,QAAQ,WAAW,qBAAqB;CACxE,MAAM,kBAAkB,WAAW,YAAY;CAC/C,MAAM,mBAAmB,WAAW,aAAa;CACjD,MAAM,WAAW,KAAK,SAAS,QAC7B,IAAI,eAAe,SAAS,eAAe,CAAC,IAAI,CAAC,IAAI,eAAe,GAAG,CACzE;CACA,MAAM,aAAa,OAAO,KAAK,UAAU,MAAM,OAAO,CAAC,CAAC,OAAO,cAAc;CAC7E,MAAM,cACJ,KAAK,SAAS,IACV,SAAS,WAAW,KAAK,SACvB,WAAW,QAAQ,IACnB,OACF,WAAW,UAAU;CAC3B,MAAM,YACJ,KAAK,SAAS,IACV,KAAK,KAAK,QAAQ,IAAI,MAAM,IAC5B,OAAO,KAAK,UAAU,MAAM,UAAU,CAAC,CAAC,OAAO,cAAc;CACnE,MAAM,UAAoC;EACxC;EACA;EACA;EACA;EACA;EACA;EACA;EACA,UAAU,gBAAgB,YAAY;EACtC,mBAAmB,KAAK,SAAS,WAAW;EAC5C,WAAW,WAAW,aAAa;EACnC;EACA;EACA,YAAY,WAAW,iBAAiB,gBAAgB;EACxD;EACA,WAAW,iBAAiB,WAAW,GAAI;EAC3C,YAAY,OAAO;EACnB,iBAAiB,OAAO,QAAQ,QAAQ,IAAI,MAAM,CAAC,CAAC;EACpD,kBAAkB,OAAO,QAAQ,MAAM,EAAE,cAAc,CAAC,CAAC,CAAC;EAC1D,iBAAiB,OAAO,QAAQ,OAAO,EAAE,aAAa,KAAK,CAAC,CAAC,CAAC;EAC9D;EACA,cAAc,aAAa,SAAS;EACpC,oBAAoB,oBAAoB,MAAM,QAAQ,WAAW,qBAAqB;EACtF,0BAA0B,yBAAyB,MAAM;CAC3D;CAEA,MAAM,SAAmC,CAAC;CAC1C,YAAY,OAAO,YAAY,SAAS,MAAM;CAC9C,aAAa,YAAY,SAAS,MAAM;CACxC,iBAAiB,SAAS,MAAM;CAChC,oBAAoB,MAAM,gBAAgB,MAAM,YAAY,SAAS,MAAM;CAC3E,iBAAiB,YAAY,SAAS,MAAM;CAC5C,gBAAgB,YAAY,SAAS,MAAM;CAE3C,MAAM,OAAO,UAAU,SAAS,YAAY,MAAM;CAClD,MAAM,SAAS,OAAO,MAAM,MAAM,EAAE,aAAa,UAAU,IACvD,SACA,OAAO,SAAS,IACd,SACA;CAEN,OAAO;EACL,QAAQ,MAAM;EACd;EACA,YAAY,MAAM,cAAc;EAChC;EACA,SAAS,WAAW,WAAW,MAAM,eAAe,MAAM,aAAa,UAAU;EACjF;EACA;EACA;EACA,SAAS,MAAM,WAAW;EAC1B,cAAc,MAAM,gBAAgB;EACpC,SAAS,cAAc,MAAM,QAAQ,QAAQ,SAAS,MAAM;CAC9D;AACF;AAEA,SAAgB,wBAAwB,OAA2D;CACjG,MAAM,YAAY,0BAA0B,KAAK;CACjD,IAAI,UAAU,WAAW,QACvB,MAAM,IAAI,kBAAkB,UAAU,OAAO;CAE/C,OAAO;AACT;AAEA,SAAS,gBACP,MACA,aACA,YACa;CACb,IAAI,aAAa,OAAO,KAAK,QAAQ,MAAM,EAAE,gBAAgB,WAAW;CACxE,IAAI,YAAY,OAAO,KAAK,QAAQ,MAAM,EAAE,gBAAgB,UAAU;CACtE,OAAO,CAAC,GAAG,IAAI;AACjB;AAEA,SAAS,qBACP,QACA,aACA,YACwB;CACxB,IAAI,aACF,OAAO,OAAO,QAAQ,MAAM,EAAE,gBAAgB,KAAA,KAAa,EAAE,gBAAgB,WAAW;CAC1F,IAAI,YACF,OAAO,OAAO,QAAQ,MAAM,EAAE,gBAAgB,KAAA,KAAa,EAAE,gBAAgB,UAAU;CACzF,OAAO,CAAC,GAAG,MAAM;AACnB;AAEA,SAAS,6BACP,OACA,OACsB;CACtB,MAAM,QAAQ;CACd,IAAI,OAAO,OAAO,OAAO,aAAa,GACpC,MAAM,IAAI,gBACR,UAAU,MAAM,2DAClB;CAEF,IAAI,MAAM,iBAAiB,KAAA,KAAa,CAAC,gBAAgB,SAAS,MAAM,YAAY,GAClF,MAAM,IAAI,gBACR,UAAU,MAAM,gCAAgC,gBAAgB,KAAK,IAAI,GAC3E;CAEF,OAAO;AACT;AAEA,SAAS,YACP,OACA,YACA,SACA,QACM;CACN,IAAI,WAAW,iBAAiB,CAAC,MAAM,YAAY,MAAM,WAAW,UAAU,OAAO,GACnF,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;CAEH,IAAI,QAAQ,gBAAgB,WAAW,kBACrC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,cAAc,qBAAqB,WAAW,iBAAiB;CACpF,CAAC;CAEH,IAAI,WAAW,kBAAkB,QAAQ,YAAY,YAAY,GAC/D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;AAEL;AAEA,SAAS,aACP,YACA,SACA,QACM;CACN,IAAI,QAAQ,aAAa,WAAW,eAClC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,WAAW,uBAAuB,WAAW,cAAc;CAChF,CAAC;CAEH,IAAI,QAAQ,eAAe,GACzB,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,aAAa;CAClC,CAAC;CAEH,IAAI,QAAQ,aAAa,QAAQ,QAAQ,cAAc,MACrD,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;CAEH,IAAI,QAAQ,aAAa,QAAQ,QAAQ,WAAW,WAAW,aAC7D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,YAAY,IAAI,QAAQ,QAAQ,EAAE,KAAK,IAAI,WAAW,WAAW,EAAE;CAC7E,CAAC;CAEH,IAAI,QAAQ,cAAc,QAAQ,QAAQ,YAAY,WAAW,cAC/D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,aAAa,IAAI,QAAQ,SAAS,EAAE,KAAK,IAAI,WAAW,YAAY,EAAE;CAChF,CAAC;AAEL;AAEA,SAAS,iBACP,SACA,QACM;CACN,IAAI,QAAQ,oBAAoB,MAC9B,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QACE,QAAQ,2BAA2B,IAC/B,GAAG,QAAQ,yBAAyB,wDACpC;CACR,CAAC;CAEH,IAAI,QAAQ,sBAAsB,GAChC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,oBAAoB;CACzC,CAAC;AAEL;AAEA,SAAS,oBACP,cACA,YACA,SACA,QACM;CACN,IAAI,WAAW,kBAAkB,QAAQ,cAAc,WAAW,gBAChE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,YAAY,wBAAwB,WAAW,eAAe;CACnF,CAAC;CAEH,IAAI,QAAQ,eAAe,QAAQ,QAAQ,aAAa,WAAW,eACjE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,sBAAsB,IAAI,QAAQ,UAAU,EAAE,KAAK,IAAI,WAAW,aAAa,EAAE;CAC3F,CAAC;CAEH,IAAI,gBAAgB,CAAC,aAAa,SAChC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM,QAAQ,aAAa,iBAAiB;EAC5C,QAAQ,aAAa;CACvB,CAAC;AAEL;AAEA,SAAS,iBACP,YACA,SACA,QACM;CACN,IAAI,CAAC,WAAW,uBAAuB;CACvC,IAAI,QAAQ,aAAa,QAAQ,iBAC/B,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,aAAa,QAAQ,gBAAgB;CAC1D,CAAC;AAEL;AAEA,SAAS,gBACP,YACA,SACA,QACM;CACN,IAAI,OAAO,SAAS,WAAW,cAAc,KAAK,QAAQ,gBAAgB,MACxE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;MACI,IAAI,QAAQ,gBAAgB,QAAQ,QAAQ,cAAc,WAAW,gBAC1E,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,eAAe,IAAI,QAAQ,WAAW,EAAE,KAAK,IAAI,WAAW,cAAc,EAAE;CACtF,CAAC;CAEH,IAAI,OAAO,SAAS,WAAW,YAAY,KAAK,QAAQ,cAAc,MACpE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;MACI,IAAI,QAAQ,cAAc,QAAQ,QAAQ,YAAY,WAAW,cACtE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,aAAa,IAAI,QAAQ,SAAS,EAAE,KAAK,IAAI,WAAW,YAAY,EAAE;CAChF,CAAC;AAEL;AAEA,SAAS,UACP,SACA,YACA,QACyB;CACzB,OAAO;EACL,KACE,UACA,QACA,QAAQ,QAAQ,gBAAgB,KAAK,IAAI,GAAG,WAAW,gBAAgB,CAAC,GACxE,GAAG,QAAQ,cAAc,sBAAsB,QAAQ,YAAY,SACrE;EACA,KACE,WACA,QACA,QAAQ,aAAa,QAAQ,QAAQ,cAAc,OAC/C,OACA,KAAK,IAAI,QAAQ,UAAU,QAAQ,SAAS,GAChD,YAAY,IAAI,QAAQ,QAAQ,EAAE,aAAa,IAAI,QAAQ,SAAS,GACtE;EACA,KACE,eACA,QACA,QAAQ,iBACR,eAAe,IAAI,QAAQ,eAAe,EAAE,oBAAoB,QAAQ,oBAAoB,gBAAgB,QAAQ,0BACtH;EACA,KACE,kBACA,QACA,SAAS,QAAQ,YAAY,WAAW,aAAa,GACrD,eAAe,QAAQ,YAAY,cAAc,IAAI,QAAQ,UAAU,GACzE;EACA,KACE,eACA,QACA,QAAQ,eAAe,IAAI,IAAI,QAAQ,kBAAkB,QAAQ,YACjE,mBAAmB,QAAQ,gBAAgB,GAAG,QAAQ,YACxD;EACA,KACE,cACA,QACA,gBAAgB,SAAS,UAAU,GACnC,eAAe,IAAI,QAAQ,WAAW,EAAE,aAAa,IAAI,QAAQ,SAAS,GAC5E;CACF;AACF;AAEA,SAAS,KACP,MACA,QACA,OACA,QACuB;CACvB,MAAM,MAAM,OAAO,QAAQ,MAAM,EAAE,SAAS,IAAI;CAMhD,OAAO;EAAE;EAAM,QALA,IAAI,MAAM,MAAM,EAAE,aAAa,UAAU,IACpD,SACA,IAAI,SAAS,IACX,SACA;EACiB,OAAO,UAAU,OAAO,OAAO,QAAQ,KAAK;EAAG;CAAO;AAC/E;AAEA,SAAS,oBAAoB,WAAqE;CAChG,MAAM,SAAuC;EAAE,OAAO;EAAG,KAAK;EAAG,MAAM;EAAG,SAAS;CAAE;CACrF,KAAK,MAAM,YAAY,WAAW,OAAO,SAAS,SAAS,QAAQ;CACnE,OAAO;AACT;AAEA,SAAS,aAAa,WAA+D;CACnF,MAAM,MAA8B,CAAC;CACrC,KAAK,MAAM,YAAY,WAAW;EAChC,MAAM,SAAS,SAAS,MAAM,UAAU,SAAS,MAAM,YAAY;EACnE,IAAI,WAAW,IAAI,WAAW,KAAK;CACrC;CACA,OAAO;AACT;AAEA,SAAS,oBACP,MACA,QACA,WACuC;CACvC,MAAM,MAA6C,CAAC;CACpD,KAAK,MAAM,OAAO,MAKhB,IAAI,eAAe,KAAK,SAAS,MAAM,OAAO;EAC5C,MAAM,eACJ,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,YACnD,IAAI,eACJ;EACN,IAAI,iBAAiB,IAAI,iBAAiB,KAAK;CACjD;CAEF,KAAK,MAAM,SAAS,QAClB,IAAI,iBAAiB,OAAO,SAAS,MAAM,OAAO;EAChD,MAAM,eACJ,MAAM,iBAAiB,KAAA,KAAa,MAAM,iBAAiB,YACvD,MAAM,eACN;EACN,IAAI,iBAAiB,IAAI,iBAAiB,KAAK;CACjD;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,QAAiE;CACjG,MAAM,MAA8B,CAAC;CACrC,KAAK,MAAM,SAAS,QAClB,KAAK,MAAM,OAAO,MAAM,OAAO,CAAC,GAAG;EACjC,MAAM,UAAU,IAAI,sBAAsB;EAC1C,IAAI,YAAY,IAAI,YAAY,KAAK;CACvC;CAEF,OAAO;AACT;AAEA,SAAS,WACP,MACA,QACA,WAC4B;CAC5B,MAAM,MAAkC,CAAC;CACzC,KAAK,MAAM,OAAO,MAChB,IAAI,eAAe,KAAK,SAAS,MAAM,OAAO;EAC5C,MAAM,YAAY,IAAI,QAAQ,IAAI;EAClC,IAAI,KAAK,EAAE,QAAQ,OAAO,cAAc,YAAY,YAAY,EAAE,CAAC;CACrE;CAEF,KAAK,MAAM,SAAS,QAClB,IAAI,iBAAiB,OAAO,SAAS,MAAM,OACzC,IAAI,KAAK,EAAE,SAAS,MAAM,KAAK,UAAU,KAAK,EAAE,CAAC;CAGrD,OAAO;AACT;AAEA,SAAS,gBAAgB,UAAsD;CAC7E,MAAM,aAAa,SAAS,QAAQ,YAAgC,YAAY,IAAI;CACpF,IAAI,WAAW,WAAW,GAAG,OAAO;CACpC,OAAO,WAAW,OAAO,OAAO,CAAC,CAAC,SAAS,WAAW;AACxD;AAEA,SAAS,eAAe,KAAgB,WAAmC;CACzE,IAAI,uBAAuB,GAAG,GAAG,OAAO;CACxC,MAAM,QAAQ,cAAc,GAAG;CAC/B,OAAO,UAAU,KAAA,IAAY,OAAO,SAAS;AAC/C;AAEA,SAAS,uBAAuB,KAAyB;CACvD,OAAO,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB;AAChE;AAEA,SAAS,iBAAiB,OAA6B,WAAmC;CACxF,IAAI,MAAM,iBAAiB,KAAA,KAAa,MAAM,iBAAiB,WAAW,OAAO;CACjF,IAAI,MAAM,OAAO,OAAO,OAAO;CAC/B,IAAI,eAAe,MAAM,KAAK,GAAG,OAAO,MAAM,SAAS;CACvD,OAAO,MAAM,OAAO,OAAO,OAAO;AACpC;AAEA,SAAS,wBACP,SACkD;CAClD,OAAO,YAAY,YAAY,YAAY,eAAe,YAAY;AACxE;AAEA,SAAS,gBAAgB,SAA4D;CACnF,IAAI,YAAY,aAAa,OAAO;CACpC,IAAI,wBAAwB,OAAO,GAAG,OAAO;AAE/C;AAEA,SAAS,UAAU,MAA4B,OAA8B;CAC3E,OAAO,KACJ,QAAQ,QAAQ,IAAI,aAAa,KAAK,CAAC,CACvC,IAAI,aAAa,CAAC,CAClB,OAAO,cAAc;AAC1B;;;;;;;AAQA,SAAS,cAAc,KAAoC;CACzD,MAAM,QAAQ,mBAAmB,KAAK,IAAI,aAAa,YAAY,YAAY,QAAQ;CACvF,OAAO,eAAe,KAAK,IAAI,QAAQ,KAAA;AACzC;AAEA,SAAS,WAAW,IAAsC;CACxD,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,OAAO,GAAG,QAAQ,KAAK,MAAM,MAAM,GAAG,CAAC,IAAI,GAAG;AAChD;AAEA,SAAS,iBAAiB,IAAuB,GAA0B;CACzE,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,MAAM,SAAS,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3C,OAAO,OAAO,KAAK,IAAI,OAAO,SAAS,GAAG,KAAK,IAAI,GAAG,KAAK,KAAK,IAAI,OAAO,MAAM,IAAI,CAAC,CAAC;AACzF;AAEA,SAAS,eAAe,OAAiC;CACvD,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK;AAC3D;AAEA,SAAS,WAAW,GAAkB,GAAiC;CACrE,IAAI,MAAM,QAAQ,MAAM,MAAM,OAAO;CACrC,OAAO,IAAI;AACb;AAEA,SAAS,SAAS,KAAoB,QAA+B;CACnE,IAAI,QAAQ,MAAM,OAAO;CACzB,IAAI,UAAU,GAAG,OAAO,OAAO,IAAI,IAAI;CACvC,OAAO,QAAQ,IAAI,KAAK,IAAI,GAAG,GAAG,IAAI,MAAM;AAC9C;AAEA,SAAS,gBACP,SACA,YACe;CACf,MAAM,OAAO,OAAO,SAAS,WAAW,cAAc,IAClD,QAAQ,gBAAgB,OACtB,OACA,QAAQ,WAAW,iBAAiB,KAAK,IAAI,QAAQ,aAAa,KAAK,CAAC,IAC1E;CACJ,MAAM,UAAU,OAAO,SAAS,WAAW,YAAY,IACnD,QAAQ,cAAc,OACpB,OACA,QAAQ,WAAW,eAAe,KAAK,IAAI,QAAQ,WAAW,KAAK,CAAC,IACtE;CACJ,IAAI,SAAS,QAAQ,YAAY,MAAM,OAAO;CAC9C,OAAO,KAAK,IAAI,MAAM,OAAO;AAC/B;AAEA,SAAS,QAAQ,GAAmB;CAClC,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;AACnC;AAEA,SAAS,cACP,QACA,QACA,SACA,QACQ;CACR,MAAM,SAAS,sBAAsB,OAAO,IAAI;CAChD,MAAM,aAAa,aAAa,QAAQ,cAAc,cAAc,QAAQ,WAAW,eAAe,QAAQ,YAAY,YAAY,IAAI,QAAQ,QAAQ,EAAE,aAAa,IAAI,QAAQ,SAAS;CAC9L,IAAI,OAAO,WAAW,GAAG,OAAO,GAAG,OAAO,IAAI;CAC9C,OAAO,GAAG,OAAO,IAAI,WAAW,WAAW,OAAO,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,GAAG;AAC/E;AAEA,SAAS,IAAI,GAA0B;CACrC,IAAI,MAAM,MAAM,OAAO;CACvB,OAAO,EAAE,QAAQ,CAAC;AACpB"}
|
|
1
|
+
{"version":3,"file":"release-confidence-DMg8n18l.js","names":[],"sources":["../src/promotion-gate.ts","../src/release-confidence.ts"],"sourcesContent":["/**\n * Bootstrap-CI promotion gate.\n *\n * In any iterative-improvement loop (GEPA, prompt evolution, dataset\n * curation), the question is \"did this generation actually improve, or are\n * we celebrating noise?\". With small N and noisy outcomes, point-estimate\n * deltas lie. Bootstrap confidence intervals tell the operator whether the\n * delta is real before code or prompts get promoted.\n *\n * This module is pure functions — no I/O, no model calls. Easy to unit-test\n * and to compose into any verdict gate.\n *\n * Default gate:\n * - Bootstrap mean baseline vs candidate (1k resamples).\n * - Compute the delta distribution; pass if the lower CI bound > 0.\n * - Tunable confidence (default 95%) and resample count.\n *\n * Verdict semantics intentionally match the existing `experiments.jsonl`\n * vocabulary:\n * - ADVANCE: candidate's CI lower bound > baseline mean (real win)\n * - KEEP: overlap, but candidate point estimate >= baseline (neutral)\n * - REVERT: candidate's CI upper bound < baseline mean (real regression)\n * - INCONCLUSIVE: not enough samples or CI straddles zero with no signal\n */\n\nimport { mulberry32 } from './statistics'\n\nexport type Verdict = 'ADVANCE' | 'KEEP' | 'REVERT' | 'INCONCLUSIVE'\n\nexport interface BootstrapResult {\n baselineMean: number\n candidateMean: number\n /** candidateMean - baselineMean, point estimate. */\n delta: number\n /** Lower bound of the (1 - alpha) CI on the delta. */\n ciLower: number\n /** Upper bound of the (1 - alpha) CI on the delta. */\n ciUpper: number\n /** Number of bootstrap resamples used. */\n iterations: number\n alpha: number\n verdict: Verdict\n}\n\nexport interface BootstrapOptions {\n /** Confidence level alpha (default 0.05 → 95% CI). */\n alpha?: number\n /** Number of resamples (default 1000). */\n iterations?: number\n /**\n * Minimum total samples (baseline + candidate) below which we always\n * return INCONCLUSIVE — bootstrap with too few samples is meaningless.\n * Default 6 (combined).\n */\n minTotalSamples?: number\n /** RNG seed for reproducibility. Absent, the seed is derived from the\n * compared arms, so the same inputs reproduce the same verdict. */\n seed?: number\n}\n\n/**\n * Compute the bootstrap CI on (candidateMean - baselineMean) and a verdict.\n *\n * Uses simple percentile bootstrap on the difference of resampled means.\n * That's the standard non-parametric primitive — no distributional\n * assumptions, robust to skew, easy to reason about.\n */\nexport function bootstrapCi(\n baseline: number[],\n candidate: number[],\n options: BootstrapOptions = {},\n): BootstrapResult {\n const alpha = options.alpha ?? 0.05\n const iterations = options.iterations ?? 1000\n const minTotal = options.minTotalSamples ?? 6\n const rng = mulberry32(options.seed ?? hashSeed(baseline, candidate))\n\n const baselineMean = mean(baseline)\n const candidateMean = mean(candidate)\n const delta = candidateMean - baselineMean\n\n if (\n baseline.length + candidate.length < minTotal ||\n baseline.length === 0 ||\n candidate.length === 0\n ) {\n return {\n baselineMean,\n candidateMean,\n delta,\n ciLower: -Infinity,\n ciUpper: Infinity,\n iterations: 0,\n alpha,\n verdict: 'INCONCLUSIVE',\n }\n }\n\n const deltas: number[] = new Array(iterations)\n for (let i = 0; i < iterations; i++) {\n const bResample = resample(baseline, rng)\n const cResample = resample(candidate, rng)\n deltas[i] = mean(cResample) - mean(bResample)\n }\n deltas.sort((a, b) => a - b)\n const lowerIdx = Math.floor((alpha / 2) * iterations)\n const upperIdx = Math.floor((1 - alpha / 2) * iterations) - 1\n const ciLower = deltas[Math.max(0, lowerIdx)]!\n const ciUpper = deltas[Math.min(iterations - 1, upperIdx)]!\n\n let verdict: Verdict\n // A ZERO-WIDTH interval is an absence of evidence, not a certainty. Constant\n // arms make every resample identical, so pass/fail data promotes on nothing:\n // [0,0,0] vs [1,1,1] is six samples, clears the `minTotalSamples` floor, and\n // yields [1, 1] — an \"ADVANCE\" carrying no information about how far the\n // estimate could be wrong. Same rule as `pairedDeltaTest`, one estimator over.\n if (!Number.isFinite(ciLower) || !Number.isFinite(ciUpper) || ciLower === ciUpper) {\n verdict = 'INCONCLUSIVE'\n } else if (ciLower > 0) verdict = 'ADVANCE'\n else if (ciUpper < 0) verdict = 'REVERT'\n else if (delta >= 0) verdict = 'KEEP'\n else verdict = 'INCONCLUSIVE'\n\n return {\n baselineMean,\n candidateMean,\n delta,\n ciLower,\n ciUpper,\n iterations,\n alpha,\n verdict,\n }\n}\n\nfunction mean(xs: number[]): number {\n if (xs.length === 0) return 0\n let s = 0\n for (const x of xs) s += x\n return s / xs.length\n}\n\nfunction resample(xs: number[], rng: () => number): number[] {\n const out = new Array(xs.length)\n for (let i = 0; i < xs.length; i++) out[i] = xs[Math.floor(rng() * xs.length)]\n return out\n}\n\n/** Stable seed derived from the inputs — same data → same CI bounds. */\nfunction hashSeed(a: number[], b: number[]): number {\n let h = 2166136261\n for (const x of [...a, ...b]) {\n const view = new Float64Array([x])\n const bytes = new Uint8Array(view.buffer)\n for (const byte of bytes) {\n h ^= byte\n h = Math.imul(h, 16777619)\n }\n }\n return h >>> 0\n}\n\n/**\n * Judge-replay promotion gate.\n *\n * The cheap inner-loop judge that drives an evolution run is by definition\n * fast and noisy. When you're about to promote a winning variant to the\n * canonical default, you want a STRONGER judge (a more expensive model, a\n * human grader, a separately-trained reward model) to confirm the win\n * generalises beyond the inner loop.\n *\n * This helper takes raw winner + baseline outputs, scores both through the\n * stronger judge, and applies `bootstrapCi`. ADVANCE means the stronger\n * judge agrees the winner is real with the configured confidence. Doesn't\n * matter what shape your \"output\" is — pass a string, an object, anything\n * the judge can read.\n */\nexport interface JudgeReplayGateArgs<TOutput> {\n baselineOutputs: TOutput[]\n candidateOutputs: TOutput[]\n /** Stronger judge — async to allow LLM calls. Return a 0..N scalar score. */\n judge: (output: TOutput) => Promise<number> | number\n alpha?: number\n iterations?: number\n /** RNG seed for reproducibility. */\n seed?: number\n /** Maximum concurrent judge calls. Default 4. */\n judgeConcurrency?: number\n}\n\n/**\n * Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.\n */\nexport async function judgeReplayGate<TOutput>(\n args: JudgeReplayGateArgs<TOutput>,\n): Promise<BootstrapResult & { baselineSamples: number; candidateSamples: number }> {\n const concurrency = args.judgeConcurrency ?? 4\n const baselineScores = await scoreAll(args.baselineOutputs, args.judge, concurrency)\n const candidateScores = await scoreAll(args.candidateOutputs, args.judge, concurrency)\n const ci = bootstrapCi(baselineScores, candidateScores, {\n ...(args.alpha !== undefined ? { alpha: args.alpha } : {}),\n ...(args.iterations !== undefined ? { iterations: args.iterations } : {}),\n ...(args.seed !== undefined ? { seed: args.seed } : {}),\n })\n return {\n ...ci,\n baselineSamples: baselineScores.length,\n candidateSamples: candidateScores.length,\n }\n}\n\nasync function scoreAll<TOutput>(\n outputs: TOutput[],\n judge: (output: TOutput) => Promise<number> | number,\n concurrency: number,\n): Promise<number[]> {\n const results: number[] = new Array(outputs.length)\n let next = 0\n async function worker(): Promise<void> {\n while (true) {\n const i = next++\n if (i >= outputs.length) return\n const v = await judge(outputs[i]!)\n results[i] = Number.isFinite(v) ? v : 0\n }\n }\n await Promise.all(Array.from({ length: Math.max(1, concurrency) }, () => worker()))\n return results\n}\n","/**\n * Release confidence gate.\n *\n * This is the production-facing composition layer over the lower-level\n * primitives:\n * - Dataset manifests prove corpus/version coverage.\n * - RunRecord rows prove reproducible search/holdout outcomes.\n * - Multi-shot trace evidence carries turn counts and ASI diagnostics.\n * - HeldOutGate decisions remain the paired promotion authority.\n *\n * The gate is intentionally pure and conservative. Missing declared evidence\n * fails closed instead of being treated as a neutral zero.\n */\n\nimport type { DatasetManifest, DatasetScenario, DatasetSplit } from './dataset'\nimport { ValidationError, VerificationError } from './errors'\nimport type { GateDecision } from './held-out-gate'\nimport { isRealnessGated, observedSplitScore } from './rollout/reward'\nimport { type RunRecord, type RunSplitTag, validateRunRecord } from './run-record'\nimport { FAILURE_CLASSES, type FailureClass } from './trace/schema'\n\n/** Severity of an actionable finding attached to a run/trace. */\nexport type AsiSeverity = 'info' | 'warning' | 'error' | 'critical'\n\n/** Actionable side-info — a diagnosed finding the loop can act on. */\nexport interface ActionableSideInfo {\n /** Stable expectation/check id when available. */\n expectationId?: string\n /** Human-readable diagnosis of what happened. */\n message: string\n severity?: AsiSeverity\n /** Concrete trace excerpt, file path, tool call, screenshot id, etc. */\n evidence?: string\n /** Prompt/tool/context surface likely responsible. */\n responsibleSurface?: string\n /** Suggested fix in natural language. */\n suggestion?: string\n /** Whether this expectation was satisfied. Defaults to false for ASI rows. */\n matched?: boolean\n metadata?: Record<string, unknown>\n}\n\nexport type ReleaseConfidenceStatus = 'pass' | 'warn' | 'fail'\nexport type ReleaseConfidenceAxisName =\n | 'corpus'\n | 'quality'\n | 'reliability'\n | 'generalization'\n | 'diagnostics'\n | 'efficiency'\n\nexport interface ReleaseTraceEvidence {\n scenarioId: string\n candidateId?: string\n split?: RunSplitTag\n score?: number\n ok?: boolean\n turnCount?: number\n costUsd?: number\n durationMs?: number\n /** Canonical task-failure class. Free-form detail belongs in ASI. */\n failureClass?: FailureClass\n asi?: ActionableSideInfo[]\n metadata?: Record<string, unknown>\n}\n\nexport interface ReleaseConfidenceThresholds {\n /** Require a Dataset manifest or explicit scenarios. Default true. */\n requireCorpus?: boolean\n minScenarioCount?: number\n minSearchRuns?: number\n minHoldoutRuns?: number\n /** Require at least one holdout scenario/run. Default true. */\n requireHoldout?: boolean\n minPassRate?: number\n minMeanScore?: number\n /** Search mean may exceed holdout mean by at most this much. */\n maxOverfitGap?: number\n maxMeanCostUsd?: number\n maxP95WallMs?: number\n /** Low-score/failed rows must carry ASI. Default true. */\n requireAsiForFailures?: boolean\n /** Score below this is considered a failure for ASI coverage. Default 0.5. */\n failureScoreThreshold?: number\n}\n\nexport interface ReleaseConfidenceInput {\n target: string\n candidateId?: string\n baselineId?: string\n dataset?: DatasetManifest\n scenarios?: readonly DatasetScenario[]\n runs?: readonly RunRecord[]\n traces?: readonly ReleaseTraceEvidence[]\n gateDecision?: GateDecision | null\n thresholds?: ReleaseConfidenceThresholds\n}\n\nexport interface ReleaseConfidenceAxis {\n name: ReleaseConfidenceAxisName\n status: ReleaseConfidenceStatus\n score: number | null\n detail: string\n}\n\nexport interface ReleaseConfidenceIssue {\n axis: ReleaseConfidenceAxisName\n severity: 'critical' | 'warning'\n code: string\n detail: string\n}\n\nexport interface ReleaseConfidenceMetrics {\n scenarioCount: number\n /** Search rows with a finite search score. */\n searchRuns: number\n /** Holdout rows with a finite holdout score. */\n holdoutRuns: number\n /** Runs with neither a split-matched score nor an explicit task failure. */\n unscoredRuns: number\n /** Run rows, or trace rows when no runs exist, with no classified terminal result. */\n unclassifiedTerminalRuns: number\n /** Run rows, or trace rows when no runs exist, that ended unsuccessfully. */\n terminalFailureRuns: number\n /** Success fraction when every run or fallback trace row has a classified result. */\n reliabilityRate: number | null\n passRate: number | null\n meanScore: number | null\n searchMeanScore: number | null\n holdoutMeanScore: number | null\n overfitGap: number | null\n meanCostUsd: number | null\n p95WallMs: number | null\n failedRows: number\n failuresWithAsi: number\n singleShotTraces: number\n multiShotTraces: number\n splitCounts: Record<DatasetSplit, number>\n domainCounts: Record<string, number>\n failureClassCounts: Partial<Record<FailureClass, number>>\n responsibleSurfaceCounts: Record<string, number>\n /**\n * Runs excluded from `passRate` because the authenticity gate flagged them as\n * gamed. Surfaced, never silent: a release whose pass rate is computed over a\n * shrunken denominator has to say by how much, or the exclusion is just a\n * different way of hiding the same runs.\n */\n realnessGatedRuns: number\n}\n\nexport interface ReleaseConfidenceScorecard {\n target: string\n candidateId: string | null\n baselineId: string | null\n status: ReleaseConfidenceStatus\n promote: boolean\n axes: ReleaseConfidenceAxis[]\n issues: ReleaseConfidenceIssue[]\n metrics: ReleaseConfidenceMetrics\n dataset: DatasetManifest | null\n gateDecision: GateDecision | null\n summary: string\n}\n\nconst DEFAULT_THRESHOLDS: Required<ReleaseConfidenceThresholds> = {\n requireCorpus: true,\n minScenarioCount: 1,\n minSearchRuns: 1,\n minHoldoutRuns: 1,\n requireHoldout: true,\n minPassRate: 0.8,\n minMeanScore: 0.7,\n maxOverfitGap: 0.15,\n maxMeanCostUsd: Number.POSITIVE_INFINITY,\n maxP95WallMs: Number.POSITIVE_INFINITY,\n requireAsiForFailures: true,\n failureScoreThreshold: 0.5,\n}\n\nexport function evaluateReleaseConfidence(\n input: ReleaseConfidenceInput,\n): ReleaseConfidenceScorecard {\n const thresholds = { ...DEFAULT_THRESHOLDS, ...input.thresholds }\n const candidateId = input.candidateId ?? null\n const runs = filterCandidate(\n (input.runs ?? []).map(validateRunRecord),\n candidateId,\n input.baselineId,\n )\n const traces = filterTraceCandidate(\n (input.traces ?? []).map(validateReleaseTraceEvidence),\n candidateId,\n input.baselineId,\n )\n const scenarios = input.scenarios ?? []\n const scenarioCount = input.dataset?.scenarioCount ?? scenarios.length\n const splitCounts = input.dataset?.splitCounts ?? countScenarioSplits(scenarios)\n const searchScores = scoresFor(runs, 'search')\n const holdoutScores = scoresFor(runs, 'holdout')\n const runScores = runs.map(runSplitScore).filter(isFiniteNumber)\n const traceScores = traces.map((t) => t.score).filter(isFiniteNumber)\n const scoreUniverse = runs.length > 0 ? runScores : traceScores\n const qualityRuns = runs.filter((run) => runSplitScore(run) !== undefined)\n const unscoredRuns = runs.filter(\n (run) =>\n runSplitScore(run) === undefined &&\n !hasExplicitTaskFailure(run) &&\n !isFailedTerminalOutcome(run.terminalOutcome),\n ).length\n // Realness-gated runs are EXCLUDED from the pass rate — numerator AND\n // denominator — and the count of what was dropped ships beside the rate as\n // `metrics.realnessGatedRuns`. Counting a gamed run as a pass made faking a\n // success the cheapest way to improve a release scorecard; scoring it 0\n // instead would be the other error, silently deflating the rate with a run\n // the gate says carries no usable verdict at all.\n const honestRuns = runs.filter((run) => !isRealnessGated(run))\n const passOutcomes =\n runs.length > 0\n ? honestRuns.map((run) => runPassOutcome(run, thresholds.failureScoreThreshold))\n : traces.map((trace) => tracePassOutcome(trace, thresholds.failureScoreThreshold))\n const reliabilityRows =\n runs.length > 0\n ? runs.map((run) => terminalSuccess(run.terminalOutcome))\n : traces.map((trace) => trace.ok)\n const unclassifiedTerminalRuns = reliabilityRows.filter((outcome) => outcome === undefined).length\n const terminalFailureRuns = reliabilityRows.filter((outcome) => outcome === false).length\n const reliabilityRate =\n reliabilityRows.length === 0 || unclassifiedTerminalRuns > 0\n ? null\n : (reliabilityRows.length - terminalFailureRuns) / reliabilityRows.length\n const searchRuns = qualityRuns.filter((r) => r.splitTag === 'search').length\n const holdoutRuns = qualityRuns.filter((r) => r.splitTag === 'holdout').length\n const failed = failedRows(runs, traces, thresholds.failureScoreThreshold)\n const searchMeanScore = meanOrNull(searchScores)\n const holdoutMeanScore = meanOrNull(holdoutScores)\n const runCosts = runs.flatMap((run) =>\n run.costProvenance.kind === 'uncaptured' ? [] : [run.costProvenance.usd],\n )\n const traceCosts = traces.map((trace) => trace.costUsd).filter(isFiniteNumber)\n const meanCostUsd =\n runs.length > 0\n ? runCosts.length === runs.length\n ? meanOrNull(runCosts)\n : null\n : meanOrNull(traceCosts)\n const wallTimes =\n runs.length > 0\n ? runs.map((run) => run.wallMs)\n : traces.map((trace) => trace.durationMs).filter(isFiniteNumber)\n const metrics: ReleaseConfidenceMetrics = {\n scenarioCount,\n searchRuns,\n holdoutRuns,\n unscoredRuns,\n unclassifiedTerminalRuns,\n terminalFailureRuns,\n reliabilityRate,\n passRate: passOutcomeRate(passOutcomes),\n realnessGatedRuns: runs.length - honestRuns.length,\n meanScore: meanOrNull(scoreUniverse),\n searchMeanScore,\n holdoutMeanScore,\n overfitGap: diffOrNull(searchMeanScore, holdoutMeanScore),\n meanCostUsd,\n p95WallMs: percentileOrNull(wallTimes, 0.95),\n failedRows: failed.length,\n failuresWithAsi: failed.filter((row) => row.hasAsi).length,\n singleShotTraces: traces.filter((t) => t.turnCount === 1).length,\n multiShotTraces: traces.filter((t) => (t.turnCount ?? 0) > 1).length,\n splitCounts,\n domainCounts: countDomains(scenarios),\n failureClassCounts: countFailureClasses(runs, traces, thresholds.failureScoreThreshold),\n responsibleSurfaceCounts: countResponsibleSurfaces(traces),\n }\n\n const issues: ReleaseConfidenceIssue[] = []\n checkCorpus(input, thresholds, metrics, issues)\n checkQuality(thresholds, metrics, issues)\n checkReliability(metrics, issues)\n checkGeneralization(input.gateDecision ?? null, thresholds, metrics, issues)\n checkDiagnostics(thresholds, metrics, issues)\n checkEfficiency(thresholds, metrics, issues)\n\n const axes = buildAxes(metrics, thresholds, issues)\n const status = issues.some((i) => i.severity === 'critical')\n ? 'fail'\n : issues.length > 0\n ? 'warn'\n : 'pass'\n\n return {\n target: input.target,\n candidateId,\n baselineId: input.baselineId ?? null,\n status,\n promote: status === 'pass' && (input.gateDecision ? input.gateDecision.promote : true),\n axes,\n issues,\n metrics,\n dataset: input.dataset ?? null,\n gateDecision: input.gateDecision ?? null,\n summary: renderSummary(input.target, status, metrics, issues),\n }\n}\n\nexport function assertReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard {\n const scorecard = evaluateReleaseConfidence(input)\n if (scorecard.status === 'fail') {\n throw new VerificationError(scorecard.summary)\n }\n return scorecard\n}\n\nfunction filterCandidate(\n runs: readonly RunRecord[],\n candidateId: string | null,\n baselineId?: string,\n): RunRecord[] {\n if (candidateId) return runs.filter((r) => r.candidateId === candidateId)\n if (baselineId) return runs.filter((r) => r.candidateId !== baselineId)\n return [...runs]\n}\n\nfunction filterTraceCandidate(\n traces: readonly ReleaseTraceEvidence[],\n candidateId: string | null,\n baselineId?: string,\n): ReleaseTraceEvidence[] {\n if (candidateId)\n return traces.filter((t) => t.candidateId === undefined || t.candidateId === candidateId)\n if (baselineId)\n return traces.filter((t) => t.candidateId === undefined || t.candidateId !== baselineId)\n return [...traces]\n}\n\nfunction validateReleaseTraceEvidence(\n trace: ReleaseTraceEvidence,\n index: number,\n): ReleaseTraceEvidence {\n const value = trace as unknown as Record<string, unknown>\n if (Object.hasOwn(value, 'failureMode')) {\n throw new ValidationError(\n `traces[${index}].failureMode is not supported; use canonical failureClass`,\n )\n }\n if (trace.failureClass !== undefined && !FAILURE_CLASSES.includes(trace.failureClass)) {\n throw new ValidationError(\n `traces[${index}].failureClass must be one of ${FAILURE_CLASSES.join(', ')}`,\n )\n }\n return trace\n}\n\nfunction checkCorpus(\n input: ReleaseConfidenceInput,\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (thresholds.requireCorpus && !input.dataset && (input.scenarios?.length ?? 0) === 0) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'missing_corpus',\n detail: 'No Dataset manifest or scenarios supplied.',\n })\n }\n if (metrics.scenarioCount < thresholds.minScenarioCount) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'few_scenarios',\n detail: `${metrics.scenarioCount} scenario(s) < min ${thresholds.minScenarioCount}.`,\n })\n }\n if (thresholds.requireHoldout && metrics.splitCounts.holdout === 0) {\n issues.push({\n axis: 'corpus',\n severity: 'critical',\n code: 'missing_holdout_split',\n detail: 'Corpus has no holdout scenarios.',\n })\n }\n}\n\nfunction checkQuality(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (metrics.searchRuns < thresholds.minSearchRuns) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'few_search_runs',\n detail: `${metrics.searchRuns} search run(s) < min ${thresholds.minSearchRuns}.`,\n })\n }\n if (metrics.unscoredRuns > 0) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'unscored_runs',\n detail: `${metrics.unscoredRuns} supplied run(s) have no task result.`,\n })\n }\n if (metrics.passRate === null || metrics.meanScore === null) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'missing_quality_scores',\n detail: 'No task-quality scores are available for pass-rate and mean-score checks.',\n })\n }\n if (metrics.passRate !== null && metrics.passRate < thresholds.minPassRate) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'low_pass_rate',\n detail: `passRate ${fmt(metrics.passRate)} < ${fmt(thresholds.minPassRate)}.`,\n })\n }\n if (metrics.meanScore !== null && metrics.meanScore < thresholds.minMeanScore) {\n issues.push({\n axis: 'quality',\n severity: 'critical',\n code: 'low_mean_score',\n detail: `meanScore ${fmt(metrics.meanScore)} < ${fmt(thresholds.minMeanScore)}.`,\n })\n }\n}\n\nfunction checkReliability(\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (metrics.reliabilityRate === null) {\n issues.push({\n axis: 'reliability',\n severity: 'critical',\n code: 'missing_reliability_evidence',\n detail:\n metrics.unclassifiedTerminalRuns > 0\n ? `${metrics.unclassifiedTerminalRuns} supplied run(s) have no classified terminal result.`\n : 'No classified terminal results are available.',\n })\n }\n if (metrics.terminalFailureRuns > 0) {\n issues.push({\n axis: 'reliability',\n severity: 'critical',\n code: 'terminal_run_failures',\n detail: `${metrics.terminalFailureRuns} run(s) ended failed, cancelled, or incomplete.`,\n })\n }\n}\n\nfunction checkGeneralization(\n gateDecision: GateDecision | null,\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (thresholds.requireHoldout && metrics.holdoutRuns < thresholds.minHoldoutRuns) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: 'few_holdout_runs',\n detail: `${metrics.holdoutRuns} holdout run(s) < min ${thresholds.minHoldoutRuns}.`,\n })\n }\n if (metrics.overfitGap !== null && metrics.overfitGap > thresholds.maxOverfitGap) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: 'overfit_gap',\n detail: `search-holdout gap ${fmt(metrics.overfitGap)} > ${fmt(thresholds.maxOverfitGap)}.`,\n })\n }\n if (gateDecision && !gateDecision.promote) {\n issues.push({\n axis: 'generalization',\n severity: 'critical',\n code: `gate_${gateDecision.rejectionCode ?? 'reject'}`,\n detail: gateDecision.reason,\n })\n }\n}\n\nfunction checkDiagnostics(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (!thresholds.requireAsiForFailures) return\n if (metrics.failedRows > metrics.failuresWithAsi) {\n issues.push({\n axis: 'diagnostics',\n severity: 'critical',\n code: 'missing_failure_asi',\n detail: `${metrics.failedRows - metrics.failuresWithAsi} failed row(s) have no actionable side information.`,\n })\n }\n}\n\nfunction checkEfficiency(\n thresholds: Required<ReleaseConfidenceThresholds>,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): void {\n if (Number.isFinite(thresholds.maxMeanCostUsd) && metrics.meanCostUsd === null) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'missing_cost',\n detail: 'A finite cost limit was configured but no cost evidence is available.',\n })\n } else if (metrics.meanCostUsd !== null && metrics.meanCostUsd > thresholds.maxMeanCostUsd) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'cost_budget',\n detail: `meanCostUsd ${fmt(metrics.meanCostUsd)} > ${fmt(thresholds.maxMeanCostUsd)}.`,\n })\n }\n if (Number.isFinite(thresholds.maxP95WallMs) && metrics.p95WallMs === null) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'missing_latency',\n detail: 'A finite latency limit was configured but no latency evidence is available.',\n })\n } else if (metrics.p95WallMs !== null && metrics.p95WallMs > thresholds.maxP95WallMs) {\n issues.push({\n axis: 'efficiency',\n severity: 'critical',\n code: 'latency_budget',\n detail: `p95WallMs ${fmt(metrics.p95WallMs)} > ${fmt(thresholds.maxP95WallMs)}.`,\n })\n }\n}\n\nfunction buildAxes(\n metrics: ReleaseConfidenceMetrics,\n thresholds: Required<ReleaseConfidenceThresholds>,\n issues: ReleaseConfidenceIssue[],\n): ReleaseConfidenceAxis[] {\n return [\n axis(\n 'corpus',\n issues,\n bounded(metrics.scenarioCount / Math.max(1, thresholds.minScenarioCount)),\n `${metrics.scenarioCount} scenarios; holdout=${metrics.splitCounts.holdout}`,\n ),\n axis(\n 'quality',\n issues,\n metrics.passRate === null || metrics.meanScore === null\n ? null\n : Math.min(metrics.passRate, metrics.meanScore),\n `passRate=${fmt(metrics.passRate)} meanScore=${fmt(metrics.meanScore)}`,\n ),\n axis(\n 'reliability',\n issues,\n metrics.reliabilityRate,\n `successRate=${fmt(metrics.reliabilityRate)} terminalFailures=${metrics.terminalFailureRuns} unclassified=${metrics.unclassifiedTerminalRuns}`,\n ),\n axis(\n 'generalization',\n issues,\n gapScore(metrics.overfitGap, thresholds.maxOverfitGap),\n `holdoutRuns=${metrics.holdoutRuns} overfitGap=${fmt(metrics.overfitGap)}`,\n ),\n axis(\n 'diagnostics',\n issues,\n metrics.failedRows === 0 ? 1 : metrics.failuresWithAsi / metrics.failedRows,\n `failuresWithAsi=${metrics.failuresWithAsi}/${metrics.failedRows}`,\n ),\n axis(\n 'efficiency',\n issues,\n efficiencyScore(metrics, thresholds),\n `meanCostUsd=${fmt(metrics.meanCostUsd)} p95WallMs=${fmt(metrics.p95WallMs)}`,\n ),\n ]\n}\n\nfunction axis(\n name: ReleaseConfidenceAxisName,\n issues: ReleaseConfidenceIssue[],\n score: number | null,\n detail: string,\n): ReleaseConfidenceAxis {\n const own = issues.filter((i) => i.axis === name)\n const status = own.some((i) => i.severity === 'critical')\n ? 'fail'\n : own.length > 0\n ? 'warn'\n : 'pass'\n return { name, status, score: score === null ? null : bounded(score), detail }\n}\n\nfunction countScenarioSplits(scenarios: readonly DatasetScenario[]): Record<DatasetSplit, number> {\n const counts: Record<DatasetSplit, number> = { train: 0, dev: 0, test: 0, holdout: 0 }\n for (const scenario of scenarios) counts[scenario.split ?? 'train']++\n return counts\n}\n\nfunction countDomains(scenarios: readonly DatasetScenario[]): Record<string, number> {\n const out: Record<string, number> = {}\n for (const scenario of scenarios) {\n const domain = scenario.tags?.domain ?? scenario.tags?.category ?? 'uncategorized'\n out[domain] = (out[domain] ?? 0) + 1\n }\n return out\n}\n\nfunction countFailureClasses(\n runs: readonly RunRecord[],\n traces: readonly ReleaseTraceEvidence[],\n threshold: number,\n): Partial<Record<FailureClass, number>> {\n const out: Partial<Record<FailureClass, number>> = {}\n for (const run of runs) {\n // Ungated: a failure-mode census counts what the runs REPORTED, which is\n // the only way a gamed run's inflated score is visible at all. It is not a\n // promotion number — `passRate` is, and that one excludes gated runs and\n // publishes the excluded count as `metrics.realnessGatedRuns`.\n if (runPassOutcome(run, threshold) === false) {\n const failureClass =\n run.failureClass !== undefined && run.failureClass !== 'success'\n ? run.failureClass\n : 'unknown'\n out[failureClass] = (out[failureClass] ?? 0) + 1\n }\n }\n for (const trace of traces) {\n if (tracePassOutcome(trace, threshold) === false) {\n const failureClass =\n trace.failureClass !== undefined && trace.failureClass !== 'success'\n ? trace.failureClass\n : 'unknown'\n out[failureClass] = (out[failureClass] ?? 0) + 1\n }\n }\n return out\n}\n\nfunction countResponsibleSurfaces(traces: readonly ReleaseTraceEvidence[]): Record<string, number> {\n const out: Record<string, number> = {}\n for (const trace of traces) {\n for (const asi of trace.asi ?? []) {\n const surface = asi.responsibleSurface ?? 'unknown'\n out[surface] = (out[surface] ?? 0) + 1\n }\n }\n return out\n}\n\nfunction failedRows(\n runs: readonly RunRecord[],\n traces: readonly ReleaseTraceEvidence[],\n threshold: number,\n): Array<{ hasAsi: boolean }> {\n const out: Array<{ hasAsi: boolean }> = []\n for (const run of runs) {\n if (runPassOutcome(run, threshold) === false) {\n const asiMetric = run.outcome.raw.asi\n out.push({ hasAsi: typeof asiMetric === 'number' && asiMetric > 0 })\n }\n }\n for (const trace of traces) {\n if (tracePassOutcome(trace, threshold) === false) {\n out.push({ hasAsi: (trace.asi?.length ?? 0) > 0 })\n }\n }\n return out\n}\n\nfunction passOutcomeRate(outcomes: readonly (boolean | null)[]): number | null {\n const classified = outcomes.filter((outcome): outcome is boolean => outcome !== null)\n if (classified.length === 0) return null\n return classified.filter(Boolean).length / classified.length\n}\n\nfunction runPassOutcome(run: RunRecord, threshold: number): boolean | null {\n if (hasExplicitTaskFailure(run)) return false\n const score = runSplitScore(run)\n return score === undefined ? null : score >= threshold\n}\n\nfunction hasExplicitTaskFailure(run: RunRecord): boolean {\n return run.failureClass !== undefined && run.failureClass !== 'success'\n}\n\nfunction tracePassOutcome(trace: ReleaseTraceEvidence, threshold: number): boolean | null {\n if (trace.failureClass !== undefined && trace.failureClass !== 'success') return false\n if (trace.ok === false) return false\n if (isFiniteNumber(trace.score)) return trace.score >= threshold\n return trace.ok === true ? true : null\n}\n\nfunction isFailedTerminalOutcome(\n outcome: RunRecord['terminalOutcome'],\n): outcome is 'failed' | 'cancelled' | 'incomplete' {\n return outcome === 'failed' || outcome === 'cancelled' || outcome === 'incomplete'\n}\n\nfunction terminalSuccess(outcome: RunRecord['terminalOutcome']): boolean | undefined {\n if (outcome === 'succeeded') return true\n if (isFailedTerminalOutcome(outcome)) return false\n return undefined\n}\n\nfunction scoresFor(runs: readonly RunRecord[], split: RunSplitTag): number[] {\n return runs\n .filter((run) => run.splitTag === split)\n .map(runSplitScore)\n .filter(isFiniteNumber)\n}\n\n/**\n * RAW and split-exact (`observedSplitScore`): this feeds the per-split means,\n * the overfit gap, and the pass threshold — descriptions of what the runs\n * reported. A gamed run inflating them is visible next to `realnessGatedRuns`,\n * and the promotion number (`passRate`) excludes gated runs entirely.\n */\nfunction runSplitScore(run: RunRecord): number | undefined {\n const score = observedSplitScore(run, run.splitTag === 'holdout' ? 'holdout' : 'search')\n return isFiniteNumber(score) ? score : undefined\n}\n\nfunction meanOrNull(xs: readonly number[]): number | null {\n if (xs.length === 0) return null\n return xs.reduce((sum, x) => sum + x, 0) / xs.length\n}\n\nfunction percentileOrNull(xs: readonly number[], p: number): number | null {\n if (xs.length === 0) return null\n const sorted = [...xs].sort((a, b) => a - b)\n return sorted[Math.min(sorted.length - 1, Math.max(0, Math.ceil(p * sorted.length) - 1))]!\n}\n\nfunction isFiniteNumber(value: unknown): value is number {\n return typeof value === 'number' && Number.isFinite(value)\n}\n\nfunction diffOrNull(a: number | null, b: number | null): number | null {\n if (a === null || b === null) return null\n return a - b\n}\n\nfunction gapScore(gap: number | null, maxGap: number): number | null {\n if (gap === null) return null\n if (maxGap <= 0) return gap <= 0 ? 1 : 0\n return bounded(1 - Math.max(0, gap) / maxGap)\n}\n\nfunction efficiencyScore(\n metrics: ReleaseConfidenceMetrics,\n thresholds: Required<ReleaseConfidenceThresholds>,\n): number | null {\n const cost = Number.isFinite(thresholds.maxMeanCostUsd)\n ? metrics.meanCostUsd === null\n ? null\n : bounded(thresholds.maxMeanCostUsd / Math.max(metrics.meanCostUsd, 1e-12))\n : 1\n const latency = Number.isFinite(thresholds.maxP95WallMs)\n ? metrics.p95WallMs === null\n ? null\n : bounded(thresholds.maxP95WallMs / Math.max(metrics.p95WallMs, 1e-12))\n : 1\n if (cost === null || latency === null) return null\n return Math.min(cost, latency)\n}\n\nfunction bounded(x: number): number {\n if (!Number.isFinite(x)) return 0\n return Math.max(0, Math.min(1, x))\n}\n\nfunction renderSummary(\n target: string,\n status: ReleaseConfidenceStatus,\n metrics: ReleaseConfidenceMetrics,\n issues: ReleaseConfidenceIssue[],\n): string {\n const prefix = `release confidence ${status}: ${target}`\n const metricText = `scenarios=${metrics.scenarioCount} searchRuns=${metrics.searchRuns} holdoutRuns=${metrics.holdoutRuns} passRate=${fmt(metrics.passRate)} meanScore=${fmt(metrics.meanScore)}`\n if (issues.length === 0) return `${prefix}; ${metricText}`\n return `${prefix}; ${metricText}; issues=${issues.map((i) => i.code).join(',')}`\n}\n\nfunction fmt(x: number | null): string {\n if (x === null) return 'n/a'\n return x.toFixed(4)\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmEA,SAAgB,YACd,UACA,WACA,UAA4B,CAAC,GACZ;CACjB,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,aAAa,QAAQ,cAAc;CACzC,MAAM,WAAW,QAAQ,mBAAmB;CAC5C,MAAM,MAAM,WAAW,QAAQ,QAAQ,SAAS,UAAU,SAAS,CAAC;CAEpE,MAAM,eAAe,KAAK,QAAQ;CAClC,MAAM,gBAAgB,KAAK,SAAS;CACpC,MAAM,QAAQ,gBAAgB;CAE9B,IACE,SAAS,SAAS,UAAU,SAAS,YACrC,SAAS,WAAW,KACpB,UAAU,WAAW,GAErB,OAAO;EACL;EACA;EACA;EACA,SAAS;EACT,SAAS;EACT,YAAY;EACZ;EACA,SAAS;CACX;CAGF,MAAM,SAAmB,IAAI,MAAM,UAAU;CAC7C,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,KAAK;EACnC,MAAM,YAAY,SAAS,UAAU,GAAG;EAExC,OAAO,KAAK,KADM,SAAS,WAAW,GACb,CAAC,IAAI,KAAK,SAAS;CAC9C;CACA,OAAO,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3B,MAAM,WAAW,KAAK,MAAO,QAAQ,IAAK,UAAU;CACpD,MAAM,WAAW,KAAK,OAAO,IAAI,QAAQ,KAAK,UAAU,IAAI;CAC5D,MAAM,UAAU,OAAO,KAAK,IAAI,GAAG,QAAQ;CAC3C,MAAM,UAAU,OAAO,KAAK,IAAI,aAAa,GAAG,QAAQ;CAExD,IAAI;CAMJ,IAAI,CAAC,OAAO,SAAS,OAAO,KAAK,CAAC,OAAO,SAAS,OAAO,KAAK,YAAY,SACxE,UAAU;MACL,IAAI,UAAU,GAAG,UAAU;MAC7B,IAAI,UAAU,GAAG,UAAU;MAC3B,IAAI,SAAS,GAAG,UAAU;MAC1B,UAAU;CAEf,OAAO;EACL;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF;AACF;AAEA,SAAS,KAAK,IAAsB;CAClC,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,IAAI,KAAK;CACzB,OAAO,IAAI,GAAG;AAChB;AAEA,SAAS,SAAS,IAAc,KAA6B;CAC3D,MAAM,MAAM,IAAI,MAAM,GAAG,MAAM;CAC/B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,QAAQ,KAAK,IAAI,KAAK,GAAG,KAAK,MAAM,IAAI,IAAI,GAAG,MAAM;CAC5E,OAAO;AACT;;AAGA,SAAS,SAAS,GAAa,GAAqB;CAClD,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,CAAC,GAAG,GAAG,GAAG,CAAC,GAAG;EAC5B,MAAM,OAAO,IAAI,aAAa,CAAC,CAAC,CAAC;EACjC,MAAM,QAAQ,IAAI,WAAW,KAAK,MAAM;EACxC,KAAK,MAAM,QAAQ,OAAO;GACxB,KAAK;GACL,IAAI,KAAK,KAAK,GAAG,QAAQ;EAC3B;CACF;CACA,OAAO,MAAM;AACf;;;;AAiCA,eAAsB,gBACpB,MACkF;CAClF,MAAM,cAAc,KAAK,oBAAoB;CAC7C,MAAM,iBAAiB,MAAM,SAAS,KAAK,iBAAiB,KAAK,OAAO,WAAW;CACnF,MAAM,kBAAkB,MAAM,SAAS,KAAK,kBAAkB,KAAK,OAAO,WAAW;CAMrF,OAAO;EACL,GANS,YAAY,gBAAgB,iBAAiB;GACtD,GAAI,KAAK,UAAU,KAAA,IAAY,EAAE,OAAO,KAAK,MAAM,IAAI,CAAC;GACxD,GAAI,KAAK,eAAe,KAAA,IAAY,EAAE,YAAY,KAAK,WAAW,IAAI,CAAC;GACvE,GAAI,KAAK,SAAS,KAAA,IAAY,EAAE,MAAM,KAAK,KAAK,IAAI,CAAC;EACvD,CAEM;EACJ,iBAAiB,eAAe;EAChC,kBAAkB,gBAAgB;CACpC;AACF;AAEA,eAAe,SACb,SACA,OACA,aACmB;CACnB,MAAM,UAAoB,IAAI,MAAM,QAAQ,MAAM;CAClD,IAAI,OAAO;CACX,eAAe,SAAwB;EACrC,OAAO,MAAM;GACX,MAAM,IAAI;GACV,IAAI,KAAK,QAAQ,QAAQ;GACzB,MAAM,IAAI,MAAM,MAAM,QAAQ,EAAG;GACjC,QAAQ,KAAK,OAAO,SAAS,CAAC,IAAI,IAAI;EACxC;CACF;CACA,MAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,GAAG,WAAW,EAAE,SAAS,OAAO,CAAC,CAAC;CAClF,OAAO;AACT;;;AChEA,MAAM,qBAA4D;CAChE,eAAe;CACf,kBAAkB;CAClB,eAAe;CACf,gBAAgB;CAChB,gBAAgB;CAChB,aAAa;CACb,cAAc;CACd,eAAe;CACf,gBAAgB,OAAO;CACvB,cAAc,OAAO;CACrB,uBAAuB;CACvB,uBAAuB;AACzB;AAEA,SAAgB,0BACd,OAC4B;CAC5B,MAAM,aAAa;EAAE,GAAG;EAAoB,GAAG,MAAM;CAAW;CAChE,MAAM,cAAc,MAAM,eAAe;CACzC,MAAM,OAAO,iBACV,MAAM,QAAQ,CAAC,EAAA,CAAG,IAAI,iBAAiB,GACxC,aACA,MAAM,UACR;CACA,MAAM,SAAS,sBACZ,MAAM,UAAU,CAAC,EAAA,CAAG,IAAI,4BAA4B,GACrD,aACA,MAAM,UACR;CACA,MAAM,YAAY,MAAM,aAAa,CAAC;CACtC,MAAM,gBAAgB,MAAM,SAAS,iBAAiB,UAAU;CAChE,MAAM,cAAc,MAAM,SAAS,eAAe,oBAAoB,SAAS;CAC/E,MAAM,eAAe,UAAU,MAAM,QAAQ;CAC7C,MAAM,gBAAgB,UAAU,MAAM,SAAS;CAC/C,MAAM,YAAY,KAAK,IAAI,aAAa,CAAC,CAAC,OAAO,cAAc;CAC/D,MAAM,cAAc,OAAO,KAAK,MAAM,EAAE,KAAK,CAAC,CAAC,OAAO,cAAc;CACpE,MAAM,gBAAgB,KAAK,SAAS,IAAI,YAAY;CACpD,MAAM,cAAc,KAAK,QAAQ,QAAQ,cAAc,GAAG,MAAM,KAAA,CAAS;CACzE,MAAM,eAAe,KAAK,QACvB,QACC,cAAc,GAAG,MAAM,KAAA,KACvB,CAAC,uBAAuB,GAAG,KAC3B,CAAC,wBAAwB,IAAI,eAAe,CAChD,CAAC,CAAC;CAOF,MAAM,aAAa,KAAK,QAAQ,QAAQ,CAAC,gBAAgB,GAAG,CAAC;CAC7D,MAAM,eACJ,KAAK,SAAS,IACV,WAAW,KAAK,QAAQ,eAAe,KAAK,WAAW,qBAAqB,CAAC,IAC7E,OAAO,KAAK,UAAU,iBAAiB,OAAO,WAAW,qBAAqB,CAAC;CACrF,MAAM,kBACJ,KAAK,SAAS,IACV,KAAK,KAAK,QAAQ,gBAAgB,IAAI,eAAe,CAAC,IACtD,OAAO,KAAK,UAAU,MAAM,EAAE;CACpC,MAAM,2BAA2B,gBAAgB,QAAQ,YAAY,YAAY,KAAA,CAAS,CAAC,CAAC;CAC5F,MAAM,sBAAsB,gBAAgB,QAAQ,YAAY,YAAY,KAAK,CAAC,CAAC;CACnF,MAAM,kBACJ,gBAAgB,WAAW,KAAK,2BAA2B,IACvD,QACC,gBAAgB,SAAS,uBAAuB,gBAAgB;CACvE,MAAM,aAAa,YAAY,QAAQ,MAAM,EAAE,aAAa,QAAQ,CAAC,CAAC;CACtE,MAAM,cAAc,YAAY,QAAQ,MAAM,EAAE,aAAa,SAAS,CAAC,CAAC;CACxE,MAAM,SAAS,WAAW,MAAM,QAAQ,WAAW,qBAAqB;CACxE,MAAM,kBAAkB,WAAW,YAAY;CAC/C,MAAM,mBAAmB,WAAW,aAAa;CACjD,MAAM,WAAW,KAAK,SAAS,QAC7B,IAAI,eAAe,SAAS,eAAe,CAAC,IAAI,CAAC,IAAI,eAAe,GAAG,CACzE;CACA,MAAM,aAAa,OAAO,KAAK,UAAU,MAAM,OAAO,CAAC,CAAC,OAAO,cAAc;CAC7E,MAAM,cACJ,KAAK,SAAS,IACV,SAAS,WAAW,KAAK,SACvB,WAAW,QAAQ,IACnB,OACF,WAAW,UAAU;CAC3B,MAAM,YACJ,KAAK,SAAS,IACV,KAAK,KAAK,QAAQ,IAAI,MAAM,IAC5B,OAAO,KAAK,UAAU,MAAM,UAAU,CAAC,CAAC,OAAO,cAAc;CACnE,MAAM,UAAoC;EACxC;EACA;EACA;EACA;EACA;EACA;EACA;EACA,UAAU,gBAAgB,YAAY;EACtC,mBAAmB,KAAK,SAAS,WAAW;EAC5C,WAAW,WAAW,aAAa;EACnC;EACA;EACA,YAAY,WAAW,iBAAiB,gBAAgB;EACxD;EACA,WAAW,iBAAiB,WAAW,GAAI;EAC3C,YAAY,OAAO;EACnB,iBAAiB,OAAO,QAAQ,QAAQ,IAAI,MAAM,CAAC,CAAC;EACpD,kBAAkB,OAAO,QAAQ,MAAM,EAAE,cAAc,CAAC,CAAC,CAAC;EAC1D,iBAAiB,OAAO,QAAQ,OAAO,EAAE,aAAa,KAAK,CAAC,CAAC,CAAC;EAC9D;EACA,cAAc,aAAa,SAAS;EACpC,oBAAoB,oBAAoB,MAAM,QAAQ,WAAW,qBAAqB;EACtF,0BAA0B,yBAAyB,MAAM;CAC3D;CAEA,MAAM,SAAmC,CAAC;CAC1C,YAAY,OAAO,YAAY,SAAS,MAAM;CAC9C,aAAa,YAAY,SAAS,MAAM;CACxC,iBAAiB,SAAS,MAAM;CAChC,oBAAoB,MAAM,gBAAgB,MAAM,YAAY,SAAS,MAAM;CAC3E,iBAAiB,YAAY,SAAS,MAAM;CAC5C,gBAAgB,YAAY,SAAS,MAAM;CAE3C,MAAM,OAAO,UAAU,SAAS,YAAY,MAAM;CAClD,MAAM,SAAS,OAAO,MAAM,MAAM,EAAE,aAAa,UAAU,IACvD,SACA,OAAO,SAAS,IACd,SACA;CAEN,OAAO;EACL,QAAQ,MAAM;EACd;EACA,YAAY,MAAM,cAAc;EAChC;EACA,SAAS,WAAW,WAAW,MAAM,eAAe,MAAM,aAAa,UAAU;EACjF;EACA;EACA;EACA,SAAS,MAAM,WAAW;EAC1B,cAAc,MAAM,gBAAgB;EACpC,SAAS,cAAc,MAAM,QAAQ,QAAQ,SAAS,MAAM;CAC9D;AACF;AAEA,SAAgB,wBAAwB,OAA2D;CACjG,MAAM,YAAY,0BAA0B,KAAK;CACjD,IAAI,UAAU,WAAW,QACvB,MAAM,IAAI,kBAAkB,UAAU,OAAO;CAE/C,OAAO;AACT;AAEA,SAAS,gBACP,MACA,aACA,YACa;CACb,IAAI,aAAa,OAAO,KAAK,QAAQ,MAAM,EAAE,gBAAgB,WAAW;CACxE,IAAI,YAAY,OAAO,KAAK,QAAQ,MAAM,EAAE,gBAAgB,UAAU;CACtE,OAAO,CAAC,GAAG,IAAI;AACjB;AAEA,SAAS,qBACP,QACA,aACA,YACwB;CACxB,IAAI,aACF,OAAO,OAAO,QAAQ,MAAM,EAAE,gBAAgB,KAAA,KAAa,EAAE,gBAAgB,WAAW;CAC1F,IAAI,YACF,OAAO,OAAO,QAAQ,MAAM,EAAE,gBAAgB,KAAA,KAAa,EAAE,gBAAgB,UAAU;CACzF,OAAO,CAAC,GAAG,MAAM;AACnB;AAEA,SAAS,6BACP,OACA,OACsB;CACtB,MAAM,QAAQ;CACd,IAAI,OAAO,OAAO,OAAO,aAAa,GACpC,MAAM,IAAI,gBACR,UAAU,MAAM,2DAClB;CAEF,IAAI,MAAM,iBAAiB,KAAA,KAAa,CAAC,gBAAgB,SAAS,MAAM,YAAY,GAClF,MAAM,IAAI,gBACR,UAAU,MAAM,gCAAgC,gBAAgB,KAAK,IAAI,GAC3E;CAEF,OAAO;AACT;AAEA,SAAS,YACP,OACA,YACA,SACA,QACM;CACN,IAAI,WAAW,iBAAiB,CAAC,MAAM,YAAY,MAAM,WAAW,UAAU,OAAO,GACnF,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;CAEH,IAAI,QAAQ,gBAAgB,WAAW,kBACrC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,cAAc,qBAAqB,WAAW,iBAAiB;CACpF,CAAC;CAEH,IAAI,WAAW,kBAAkB,QAAQ,YAAY,YAAY,GAC/D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;AAEL;AAEA,SAAS,aACP,YACA,SACA,QACM;CACN,IAAI,QAAQ,aAAa,WAAW,eAClC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,WAAW,uBAAuB,WAAW,cAAc;CAChF,CAAC;CAEH,IAAI,QAAQ,eAAe,GACzB,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,aAAa;CAClC,CAAC;CAEH,IAAI,QAAQ,aAAa,QAAQ,QAAQ,cAAc,MACrD,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;CAEH,IAAI,QAAQ,aAAa,QAAQ,QAAQ,WAAW,WAAW,aAC7D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,YAAY,IAAI,QAAQ,QAAQ,EAAE,KAAK,IAAI,WAAW,WAAW,EAAE;CAC7E,CAAC;CAEH,IAAI,QAAQ,cAAc,QAAQ,QAAQ,YAAY,WAAW,cAC/D,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,aAAa,IAAI,QAAQ,SAAS,EAAE,KAAK,IAAI,WAAW,YAAY,EAAE;CAChF,CAAC;AAEL;AAEA,SAAS,iBACP,SACA,QACM;CACN,IAAI,QAAQ,oBAAoB,MAC9B,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QACE,QAAQ,2BAA2B,IAC/B,GAAG,QAAQ,yBAAyB,wDACpC;CACR,CAAC;CAEH,IAAI,QAAQ,sBAAsB,GAChC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,oBAAoB;CACzC,CAAC;AAEL;AAEA,SAAS,oBACP,cACA,YACA,SACA,QACM;CACN,IAAI,WAAW,kBAAkB,QAAQ,cAAc,WAAW,gBAChE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,YAAY,wBAAwB,WAAW,eAAe;CACnF,CAAC;CAEH,IAAI,QAAQ,eAAe,QAAQ,QAAQ,aAAa,WAAW,eACjE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,sBAAsB,IAAI,QAAQ,UAAU,EAAE,KAAK,IAAI,WAAW,aAAa,EAAE;CAC3F,CAAC;CAEH,IAAI,gBAAgB,CAAC,aAAa,SAChC,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM,QAAQ,aAAa,iBAAiB;EAC5C,QAAQ,aAAa;CACvB,CAAC;AAEL;AAEA,SAAS,iBACP,YACA,SACA,QACM;CACN,IAAI,CAAC,WAAW,uBAAuB;CACvC,IAAI,QAAQ,aAAa,QAAQ,iBAC/B,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,GAAG,QAAQ,aAAa,QAAQ,gBAAgB;CAC1D,CAAC;AAEL;AAEA,SAAS,gBACP,YACA,SACA,QACM;CACN,IAAI,OAAO,SAAS,WAAW,cAAc,KAAK,QAAQ,gBAAgB,MACxE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;MACI,IAAI,QAAQ,gBAAgB,QAAQ,QAAQ,cAAc,WAAW,gBAC1E,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,eAAe,IAAI,QAAQ,WAAW,EAAE,KAAK,IAAI,WAAW,cAAc,EAAE;CACtF,CAAC;CAEH,IAAI,OAAO,SAAS,WAAW,YAAY,KAAK,QAAQ,cAAc,MACpE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ;CACV,CAAC;MACI,IAAI,QAAQ,cAAc,QAAQ,QAAQ,YAAY,WAAW,cACtE,OAAO,KAAK;EACV,MAAM;EACN,UAAU;EACV,MAAM;EACN,QAAQ,aAAa,IAAI,QAAQ,SAAS,EAAE,KAAK,IAAI,WAAW,YAAY,EAAE;CAChF,CAAC;AAEL;AAEA,SAAS,UACP,SACA,YACA,QACyB;CACzB,OAAO;EACL,KACE,UACA,QACA,QAAQ,QAAQ,gBAAgB,KAAK,IAAI,GAAG,WAAW,gBAAgB,CAAC,GACxE,GAAG,QAAQ,cAAc,sBAAsB,QAAQ,YAAY,SACrE;EACA,KACE,WACA,QACA,QAAQ,aAAa,QAAQ,QAAQ,cAAc,OAC/C,OACA,KAAK,IAAI,QAAQ,UAAU,QAAQ,SAAS,GAChD,YAAY,IAAI,QAAQ,QAAQ,EAAE,aAAa,IAAI,QAAQ,SAAS,GACtE;EACA,KACE,eACA,QACA,QAAQ,iBACR,eAAe,IAAI,QAAQ,eAAe,EAAE,oBAAoB,QAAQ,oBAAoB,gBAAgB,QAAQ,0BACtH;EACA,KACE,kBACA,QACA,SAAS,QAAQ,YAAY,WAAW,aAAa,GACrD,eAAe,QAAQ,YAAY,cAAc,IAAI,QAAQ,UAAU,GACzE;EACA,KACE,eACA,QACA,QAAQ,eAAe,IAAI,IAAI,QAAQ,kBAAkB,QAAQ,YACjE,mBAAmB,QAAQ,gBAAgB,GAAG,QAAQ,YACxD;EACA,KACE,cACA,QACA,gBAAgB,SAAS,UAAU,GACnC,eAAe,IAAI,QAAQ,WAAW,EAAE,aAAa,IAAI,QAAQ,SAAS,GAC5E;CACF;AACF;AAEA,SAAS,KACP,MACA,QACA,OACA,QACuB;CACvB,MAAM,MAAM,OAAO,QAAQ,MAAM,EAAE,SAAS,IAAI;CAMhD,OAAO;EAAE;EAAM,QALA,IAAI,MAAM,MAAM,EAAE,aAAa,UAAU,IACpD,SACA,IAAI,SAAS,IACX,SACA;EACiB,OAAO,UAAU,OAAO,OAAO,QAAQ,KAAK;EAAG;CAAO;AAC/E;AAEA,SAAS,oBAAoB,WAAqE;CAChG,MAAM,SAAuC;EAAE,OAAO;EAAG,KAAK;EAAG,MAAM;EAAG,SAAS;CAAE;CACrF,KAAK,MAAM,YAAY,WAAW,OAAO,SAAS,SAAS,QAAQ;CACnE,OAAO;AACT;AAEA,SAAS,aAAa,WAA+D;CACnF,MAAM,MAA8B,CAAC;CACrC,KAAK,MAAM,YAAY,WAAW;EAChC,MAAM,SAAS,SAAS,MAAM,UAAU,SAAS,MAAM,YAAY;EACnE,IAAI,WAAW,IAAI,WAAW,KAAK;CACrC;CACA,OAAO;AACT;AAEA,SAAS,oBACP,MACA,QACA,WACuC;CACvC,MAAM,MAA6C,CAAC;CACpD,KAAK,MAAM,OAAO,MAKhB,IAAI,eAAe,KAAK,SAAS,MAAM,OAAO;EAC5C,MAAM,eACJ,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,YACnD,IAAI,eACJ;EACN,IAAI,iBAAiB,IAAI,iBAAiB,KAAK;CACjD;CAEF,KAAK,MAAM,SAAS,QAClB,IAAI,iBAAiB,OAAO,SAAS,MAAM,OAAO;EAChD,MAAM,eACJ,MAAM,iBAAiB,KAAA,KAAa,MAAM,iBAAiB,YACvD,MAAM,eACN;EACN,IAAI,iBAAiB,IAAI,iBAAiB,KAAK;CACjD;CAEF,OAAO;AACT;AAEA,SAAS,yBAAyB,QAAiE;CACjG,MAAM,MAA8B,CAAC;CACrC,KAAK,MAAM,SAAS,QAClB,KAAK,MAAM,OAAO,MAAM,OAAO,CAAC,GAAG;EACjC,MAAM,UAAU,IAAI,sBAAsB;EAC1C,IAAI,YAAY,IAAI,YAAY,KAAK;CACvC;CAEF,OAAO;AACT;AAEA,SAAS,WACP,MACA,QACA,WAC4B;CAC5B,MAAM,MAAkC,CAAC;CACzC,KAAK,MAAM,OAAO,MAChB,IAAI,eAAe,KAAK,SAAS,MAAM,OAAO;EAC5C,MAAM,YAAY,IAAI,QAAQ,IAAI;EAClC,IAAI,KAAK,EAAE,QAAQ,OAAO,cAAc,YAAY,YAAY,EAAE,CAAC;CACrE;CAEF,KAAK,MAAM,SAAS,QAClB,IAAI,iBAAiB,OAAO,SAAS,MAAM,OACzC,IAAI,KAAK,EAAE,SAAS,MAAM,KAAK,UAAU,KAAK,EAAE,CAAC;CAGrD,OAAO;AACT;AAEA,SAAS,gBAAgB,UAAsD;CAC7E,MAAM,aAAa,SAAS,QAAQ,YAAgC,YAAY,IAAI;CACpF,IAAI,WAAW,WAAW,GAAG,OAAO;CACpC,OAAO,WAAW,OAAO,OAAO,CAAC,CAAC,SAAS,WAAW;AACxD;AAEA,SAAS,eAAe,KAAgB,WAAmC;CACzE,IAAI,uBAAuB,GAAG,GAAG,OAAO;CACxC,MAAM,QAAQ,cAAc,GAAG;CAC/B,OAAO,UAAU,KAAA,IAAY,OAAO,SAAS;AAC/C;AAEA,SAAS,uBAAuB,KAAyB;CACvD,OAAO,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB;AAChE;AAEA,SAAS,iBAAiB,OAA6B,WAAmC;CACxF,IAAI,MAAM,iBAAiB,KAAA,KAAa,MAAM,iBAAiB,WAAW,OAAO;CACjF,IAAI,MAAM,OAAO,OAAO,OAAO;CAC/B,IAAI,eAAe,MAAM,KAAK,GAAG,OAAO,MAAM,SAAS;CACvD,OAAO,MAAM,OAAO,OAAO,OAAO;AACpC;AAEA,SAAS,wBACP,SACkD;CAClD,OAAO,YAAY,YAAY,YAAY,eAAe,YAAY;AACxE;AAEA,SAAS,gBAAgB,SAA4D;CACnF,IAAI,YAAY,aAAa,OAAO;CACpC,IAAI,wBAAwB,OAAO,GAAG,OAAO;AAE/C;AAEA,SAAS,UAAU,MAA4B,OAA8B;CAC3E,OAAO,KACJ,QAAQ,QAAQ,IAAI,aAAa,KAAK,CAAC,CACvC,IAAI,aAAa,CAAC,CAClB,OAAO,cAAc;AAC1B;;;;;;;AAQA,SAAS,cAAc,KAAoC;CACzD,MAAM,QAAQ,mBAAmB,KAAK,IAAI,aAAa,YAAY,YAAY,QAAQ;CACvF,OAAO,eAAe,KAAK,IAAI,QAAQ,KAAA;AACzC;AAEA,SAAS,WAAW,IAAsC;CACxD,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,OAAO,GAAG,QAAQ,KAAK,MAAM,MAAM,GAAG,CAAC,IAAI,GAAG;AAChD;AAEA,SAAS,iBAAiB,IAAuB,GAA0B;CACzE,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,MAAM,SAAS,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3C,OAAO,OAAO,KAAK,IAAI,OAAO,SAAS,GAAG,KAAK,IAAI,GAAG,KAAK,KAAK,IAAI,OAAO,MAAM,IAAI,CAAC,CAAC;AACzF;AAEA,SAAS,eAAe,OAAiC;CACvD,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK;AAC3D;AAEA,SAAS,WAAW,GAAkB,GAAiC;CACrE,IAAI,MAAM,QAAQ,MAAM,MAAM,OAAO;CACrC,OAAO,IAAI;AACb;AAEA,SAAS,SAAS,KAAoB,QAA+B;CACnE,IAAI,QAAQ,MAAM,OAAO;CACzB,IAAI,UAAU,GAAG,OAAO,OAAO,IAAI,IAAI;CACvC,OAAO,QAAQ,IAAI,KAAK,IAAI,GAAG,GAAG,IAAI,MAAM;AAC9C;AAEA,SAAS,gBACP,SACA,YACe;CACf,MAAM,OAAO,OAAO,SAAS,WAAW,cAAc,IAClD,QAAQ,gBAAgB,OACtB,OACA,QAAQ,WAAW,iBAAiB,KAAK,IAAI,QAAQ,aAAa,KAAK,CAAC,IAC1E;CACJ,MAAM,UAAU,OAAO,SAAS,WAAW,YAAY,IACnD,QAAQ,cAAc,OACpB,OACA,QAAQ,WAAW,eAAe,KAAK,IAAI,QAAQ,WAAW,KAAK,CAAC,IACtE;CACJ,IAAI,SAAS,QAAQ,YAAY,MAAM,OAAO;CAC9C,OAAO,KAAK,IAAI,MAAM,OAAO;AAC/B;AAEA,SAAS,QAAQ,GAAmB;CAClC,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG,OAAO;CAChC,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;AACnC;AAEA,SAAS,cACP,QACA,QACA,SACA,QACQ;CACR,MAAM,SAAS,sBAAsB,OAAO,IAAI;CAChD,MAAM,aAAa,aAAa,QAAQ,cAAc,cAAc,QAAQ,WAAW,eAAe,QAAQ,YAAY,YAAY,IAAI,QAAQ,QAAQ,EAAE,aAAa,IAAI,QAAQ,SAAS;CAC9L,IAAI,OAAO,WAAW,GAAG,OAAO,GAAG,OAAO,IAAI;CAC9C,OAAO,GAAG,OAAO,IAAI,WAAW,WAAW,OAAO,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,GAAG;AAC/E;AAEA,SAAS,IAAI,GAA0B;CACrC,IAAI,MAAM,MAAM,OAAO;CACvB,OAAO,EAAE,QAAQ,CAAC;AACpB"}
|
package/dist/reporting.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { t as benjaminiHochberg } from "./multiplicity-DIWHvysC.js";
|
|
2
|
-
import { d as PairedBootstrapResult, h as pairedBootstrap, u as PairedBootstrapOptions } from "./paired-promotion-decision-
|
|
3
|
-
import { _ as bootstrapCi, a as ReleaseConfidenceIssue, c as ReleaseConfidenceStatus, d as assertReleaseConfidence, f as evaluateReleaseConfidence, g as Verdict, h as JudgeReplayGateArgs, i as ReleaseConfidenceInput, k as wilcoxonSignedRank, l as ReleaseConfidenceThresholds, m as BootstrapResult, n as ReleaseConfidenceAxis, o as ReleaseConfidenceMetrics, p as BootstrapOptions, r as ReleaseConfidenceAxisName, s as ReleaseConfidenceScorecard, u as ReleaseTraceEvidence, v as judgeReplayGate } from "./release-confidence-
|
|
4
|
-
import { _ as paretoChart, a as ParetoPoint, c as ResearchReportCandidate, d as ResearchReportOptions, f as ResearchReportRecommendation, g as gainHistogram, h as SummaryTableRow, i as ParetoFigureSpec, l as ResearchReportDecision, m as SummaryTableOptions, n as GainDistributionFigureSpec, o as RESEARCH_REPORT_HARD_PAIR_FLOOR, p as SummaryTable, r as GainDistributionOptions, s as ResearchReport, t as GainDistributionBin, u as ResearchReportMethodology, v as researchReport, y as summaryTable } from "./summary-report-
|
|
2
|
+
import { d as PairedBootstrapResult, h as pairedBootstrap, u as PairedBootstrapOptions } from "./paired-promotion-decision-DPsMQm-0.js";
|
|
3
|
+
import { _ as bootstrapCi, a as ReleaseConfidenceIssue, c as ReleaseConfidenceStatus, d as assertReleaseConfidence, f as evaluateReleaseConfidence, g as Verdict, h as JudgeReplayGateArgs, i as ReleaseConfidenceInput, k as wilcoxonSignedRank, l as ReleaseConfidenceThresholds, m as BootstrapResult, n as ReleaseConfidenceAxis, o as ReleaseConfidenceMetrics, p as BootstrapOptions, r as ReleaseConfidenceAxisName, s as ReleaseConfidenceScorecard, u as ReleaseTraceEvidence, v as judgeReplayGate } from "./release-confidence-BcqeQTHW.js";
|
|
4
|
+
import { _ as paretoChart, a as ParetoPoint, c as ResearchReportCandidate, d as ResearchReportOptions, f as ResearchReportRecommendation, g as gainHistogram, h as SummaryTableRow, i as ParetoFigureSpec, l as ResearchReportDecision, m as SummaryTableOptions, n as GainDistributionFigureSpec, o as RESEARCH_REPORT_HARD_PAIR_FLOOR, p as SummaryTable, r as GainDistributionOptions, s as ResearchReport, t as GainDistributionBin, u as ResearchReportMethodology, v as researchReport, y as summaryTable } from "./summary-report-D1h4dlrK.js";
|
|
5
5
|
import { a as PairedEvalueStep, c as SequentialDecision, i as PairedEvalueSequence, l as evaluateInterimReleaseConfidence, n as InterimReleaseConfidenceInput, r as PairedEvalueOptions, t as InterimReleaseConfidence, u as pairedEvalueSequence } from "./sequential-BhsrMupG.js";
|
|
6
|
-
import { a as
|
|
6
|
+
import { a as RubricPredictiveValidityReport, i as RubricPredictiveValidityInput, o as RubricRanking, r as RubricOutcomePair, s as rubricPredictiveValidity } from "./rubric-predictive-validity-Bmj2_cll.js";
|
|
7
7
|
export { type BootstrapOptions, type BootstrapResult, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeReplayGateArgs, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type ParetoFigureSpec, type ParetoPoint, RESEARCH_REPORT_HARD_PAIR_FLOOR, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type RubricOutcomePair, type RubricPredictiveValidityInput, type RubricPredictiveValidityReport, type RubricRanking, type SequentialDecision, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type Verdict, assertReleaseConfidence, benjaminiHochberg, bootstrapCi, evaluateInterimReleaseConfidence, evaluateReleaseConfidence, gainHistogram, judgeReplayGate, pairedBootstrap, pairedEvalueSequence, paretoChart, researchReport, rubricPredictiveValidity, summaryTable, wilcoxonSignedRank };
|
package/dist/reporting.js
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { o as benjaminiHochberg } from "./power-and-mde-B8F2RdcD.js";
|
|
2
2
|
import { l as wilcoxonSignedRank } from "./paired-arms-D4aeIHUy.js";
|
|
3
3
|
import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
|
|
4
|
-
import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-
|
|
4
|
+
import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-e-MaOAHV.js";
|
|
5
5
|
import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-CzK5DarL.js";
|
|
6
|
-
import { i as judgeReplayGate, n as evaluateReleaseConfidence, r as bootstrapCi, t as assertReleaseConfidence } from "./release-confidence-
|
|
7
|
-
import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-
|
|
6
|
+
import { i as judgeReplayGate, n as evaluateReleaseConfidence, r as bootstrapCi, t as assertReleaseConfidence } from "./release-confidence-DMg8n18l.js";
|
|
7
|
+
import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-CCK-1B7w.js";
|
|
8
8
|
export { RESEARCH_REPORT_HARD_PAIR_FLOOR, assertReleaseConfidence, benjaminiHochberg, bootstrapCi, evaluateInterimReleaseConfidence, evaluateReleaseConfidence, gainHistogram, judgeReplayGate, pairedBootstrap, pairedEvalueSequence, paretoChart, researchReport, rubricPredictiveValidity, summaryTable, wilcoxonSignedRank };
|
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
import { i as AgentProfileCellInput, r as AgentProfileCell } from "./agent-profile-cell-
|
|
2
|
-
import { a as RunRecord, c as RunTaskFailure, n as RunCostProvenance, r as RunJudgeMetadata, s as RunSplitTag, t as JudgeScoresRecord, u as RunTokenUsage } from "./run-record-
|
|
3
|
-
import {
|
|
1
|
+
import { i as AgentProfileCellInput, r as AgentProfileCell } from "./agent-profile-cell-s__adRnK.js";
|
|
2
|
+
import { a as RunRecord, c as RunTaskFailure, n as RunCostProvenance, r as RunJudgeMetadata, s as RunSplitTag, t as JudgeScoresRecord, u as RunTokenUsage } from "./run-record-BiTWauyO.js";
|
|
3
|
+
import { rt as RawProviderSink, x as ChatClient } from "./types-CBbLtr2J.js";
|
|
4
4
|
import { s as TraceStore } from "./store-BErPvYBr.js";
|
|
5
|
-
import { b as GateDecision, d as ResearchReportOptions, s as ResearchReport } from "./summary-report-
|
|
5
|
+
import { b as GateDecision, d as ResearchReportOptions, s as ResearchReport } from "./summary-report-D1h4dlrK.js";
|
|
6
6
|
import { i as TraceEmitter, t as RunCompleteHook } from "./emitter-Cs0egaFd.js";
|
|
7
|
-
import { a as RunIntegrityReport, n as RunIntegrityExpectations } from "./integrity-
|
|
7
|
+
import { a as RunIntegrityReport, n as RunIntegrityExpectations } from "./integrity-rGOfSUle.js";
|
|
8
8
|
//#region src/eval-campaign.d.ts
|
|
9
9
|
interface CampaignVariant<V> {
|
|
10
10
|
id: string;
|
|
@@ -290,4 +290,4 @@ interface Researcher {
|
|
|
290
290
|
}
|
|
291
291
|
//#endregion
|
|
292
292
|
export { SteeringChange as a, CampaignRunContext as c, CampaignVariant as d, EvalCampaignOptions as f, runEvalCampaign as h, Researcher as i, CampaignRunOutcome as l, FailedRun as m, ExperimentResult as n, CampaignFactoryParams as o, EvalCampaignResult as p, FailureMode as r, CampaignIntegrityPolicy as s, ExperimentPlan as t, CampaignScenario as u };
|
|
293
|
-
//# sourceMappingURL=researcher-
|
|
293
|
+
//# sourceMappingURL=researcher-64T49THL.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"researcher-
|
|
1
|
+
{"version":3,"file":"researcher-64T49THL.d.ts","names":[],"sources":["../src/eval-campaign.ts","../src/researcher.ts"],"mappings":";;;;;;;;UAuEiB,gBAAgB;EAC/B;EACA,SAAS;;UAGM;EACf;;EAEA,OAAO;;UAGQ,mBAAmB;;EAElC;;EAEA;EACA,SAAS;EACT;EACA;EACA,cAAc;EACd;EACA,UAAU;;;;;;;EAOV,SAAS;EACT,OAAO;EACP,SAAS;;;;;EAKT,MAAM;;;UAIS;;;;;;EAMf,SAAS;EACT;;UAGQ;;EAER;;EAEA;;EAEA;;EAEA,gBAAgB;EAChB,YAAY;;EAEZ;;EAEA;;EAEA;;EAEA,MAAM;;EAEN,gBAAgB;;;;;;EAMhB,cAAc;;;;;;EAMd,eAAe,mBAAmB;;;KAIxB,qBAAqB,2BAA2B;KAEhD,eAAe,MAAM,KAAK,mBAAmB,OAAO,QAAQ;KAE5D;UAEK,oBAAoB;;;;;EAKnC;EACA,UAAU,gBAAgB;EAC1B,WAAW;;EAEX;;EAEA,WAAW;;EAEX;;;;;;;;EAQA,cAAc,QAAQ,uBAAuB;;;;;;;EAO7C;;;;;;EAMA,eAAe,QAAQ,0BAA0B;;;;;;;EAOjD,kBAAkB,QAAQ,0BAA0B;;;;;EAKpD;;;;;EAKA,gBAAgB;;;;;;EAMhB,YAAY;;EAEZ,qBAAqB;;;;;EAKrB,QAAQ,eAAe;;;;;EAKvB;IAAW;MAAwB,KACjC;;;;;EAOF;;EAEA;;;;EAIA;;EAEA,SAAS,QAAQ;;;;;;;EAOjB,eACI,mBACA,0BAEE,QAAQ;IACN,SAAS;IACT,cAAc;QAGd,mBACA,wBACA,QAAQ,mBAAmB;;UAGpB;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;;EAEA;EACA;;EAEA,MAAM;;EAEN,kBAAkB;EAClB,YAAY;;EAEZ,SAAS;EACT;EACA;;iBAWoB,gBAAgB,GACpC,MAAM,oBAAoB,KACzB,QAAQ;;;;UClRM;;;EAGf;;EAEA;EACA;;;IAGE;;IAEA;;;;UAKa;EACf;;;;EAIA;;;EAGA;;EAEA;;;UAIe;EACf;EACA;EACA,SAAS;;;EAGT;;EAEA;IAAU;IAAkB;;;;UAIb;EACf,MAAM;EACN,MAAM;EACN,cAAc;;;;;;;;;;;;;;;;;;;UAoBC;EACf,gBAAgB,MAAM,cAAc,QAAQ;EAC5C,cAAc,UAAU,gBAAgB,QAAQ;EAChD,YAAY,SAAS,kBAAkB,UAAU,iBAAiB,QAAQ;EAC1E,eAAe,MAAM,iBAAiB,QAAQ"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { a as RunRecord } from "./run-record-
|
|
1
|
+
import { a as RunRecord } from "./run-record-BiTWauyO.js";
|
|
2
2
|
import { x as VerificationStrategySource } from "./verdict-E4eRNf7-.js";
|
|
3
3
|
import { s as VerificationReport } from "./multi-layer-verifier-BUaQ4C17.js";
|
|
4
4
|
//#region src/rl/verifiable-reward.d.ts
|
|
@@ -251,4 +251,4 @@ interface DetectRewardHackingInput {
|
|
|
251
251
|
declare function detectRewardHacking(input: DetectRewardHackingInput): RewardHackingReport;
|
|
252
252
|
//#endregion
|
|
253
253
|
export { detectRewardHacking as a, VerifiableRewardSource as c, filterDeterministicallyRewarded as d, RewardHackingSignal as i, extractVerifiableReward as l, RewardHackingFinding as n, VerifiableReward as o, RewardHackingReport as r, VerifiableRewardExtractionOptions as s, DetectRewardHackingInput as t, extractVerifiableRewardsFromRecords as u };
|
|
254
|
-
//# sourceMappingURL=reward-hacking-
|
|
254
|
+
//# sourceMappingURL=reward-hacking-uzO_ihep.d.ts.map
|