@tangle-network/agent-eval 0.179.0 → 0.181.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/README.md +119 -146
- package/dist/adapters/http.d.ts +2 -2
- package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
- package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
- package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
- package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
- package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
- package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +4 -4
- package/dist/ast-CP9ae9B0.js +557 -0
- package/dist/ast-CP9ae9B0.js.map +1 -0
- package/dist/ast-hI-vjW6J.d.ts +457 -0
- package/dist/ast-hI-vjW6J.d.ts.map +1 -0
- package/dist/{benchmark-command-CY6Dg5t5.js → benchmark-command-B57n9vjz.js} +7 -6
- package/dist/{benchmark-command-CY6Dg5t5.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
- package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
- package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
- package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
- package/dist/cli.js +5 -8
- package/dist/cli.js.map +1 -1
- package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
- package/dist/client-CXE-U1SA.js.map +1 -0
- package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
- package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +11 -10
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
- package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
- package/dist/{default-registry-BryMEmr8.js → default-registry-aL7xUrUz.js} +2 -2
- package/dist/{default-registry-BryMEmr8.js.map → default-registry-aL7xUrUz.js.map} +1 -1
- package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
- package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
- package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
- package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
- package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
- package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
- package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
- package/dist/engine-DS1cysJy.d.ts.map +1 -0
- package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
- package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
- package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
- package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +27 -477
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +95 -559
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
- package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
- package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
- package/dist/index-Bp_6sj3x.d.ts.map +1 -0
- package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
- package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
- package/dist/{index-CbLmrWCa.d.ts → index-DNntP4ch.d.ts} +8 -8
- package/dist/{index-CbLmrWCa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
- package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
- package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
- package/dist/index.d.ts +28 -28
- package/dist/index.js +25 -16
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
- package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
- package/dist/{integrity-DsHWCebQ.js → integrity-DH5ng72x.js} +2 -2
- package/dist/{integrity-DsHWCebQ.js.map → integrity-DH5ng72x.js.map} +1 -1
- package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
- package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
- package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
- package/dist/journal-Cs9f7385.js.map +1 -0
- package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
- package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
- package/dist/llm-judge-DEFZeSiu.js.map +1 -0
- package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
- package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +138 -7
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +245 -97
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
- package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/outcome-store-BXlkwMPR.js +131 -0
- package/dist/outcome-store-BXlkwMPR.js.map +1 -0
- package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
- package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
- package/dist/pipelines/index.js +1 -1
- package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
- package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
- package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
- package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
- package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
- package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
- package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
- package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
- package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
- package/dist/{report-command-DKlXfU5r.js → report-command-V1ecVgAv.js} +27 -3
- package/dist/report-command-V1ecVgAv.js.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +3 -3
- package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
- package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
- package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
- package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
- package/dist/rl.d.ts +53 -99
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +182 -169
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
- package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
- package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
- package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
- package/dist/run-record-Br-Yzt_k.js +464 -0
- package/dist/run-record-Br-Yzt_k.js.map +1 -0
- package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
- package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
- package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
- package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
- package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
- package/dist/sequential-DAsyV2T9.js.map +1 -0
- package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
- package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
- package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
- package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
- package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
- package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
- package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
- package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
- package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
- package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
- package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +4 -2
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +3 -3
- package/dist/{terminal-record-Ce9_UjRz.js → terminal-record-BtPwKTSr.js} +58 -26
- package/dist/terminal-record-BtPwKTSr.js.map +1 -0
- package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
- package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
- package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
- package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
- package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
- package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
- package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
- package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
- package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
- package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
- package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
- package/dist/{types-vUdAx2Cj.d.ts → types-lPkDQNqJ.d.ts} +20 -2
- package/dist/{types-vUdAx2Cj.d.ts.map → types-lPkDQNqJ.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/docs/adapters-observability.md +14 -0
- package/docs/campaign-proposers.md +86 -128
- package/docs/charter.md +108 -112
- package/docs/concepts.md +157 -69
- package/docs/design/mlbenchmarks-book-review.md +440 -0
- package/docs/design/mlbenchmarks-review/observations.json +713 -0
- package/docs/design/mlbenchmarks-review/probes.mts +476 -0
- package/docs/design/mlbenchmarks-review/sources.json +200 -0
- package/docs/design/self-improvement-evidence-audit.md +263 -0
- package/docs/design.md +2 -1
- package/docs/eval-surface-map.md +95 -42
- package/docs/evaluation-integrity.md +220 -0
- package/docs/experiment.md +111 -55
- package/docs/feature-guide.md +5 -6
- package/docs/hosted-ingest-spec.md +4 -11
- package/docs/insight-report.md +187 -455
- package/docs/outcome-validity.md +182 -0
- package/docs/product-eval-adoption.md +1 -2
- package/docs/research-report-methodology.md +7 -7
- package/docs/search-history-receipts.md +8 -0
- package/docs/statistical-evidence.md +129 -0
- package/docs/verdicts.md +76 -49
- package/package.json +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
- package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
- package/dist/client-BlLY6o2w.js.map +0 -1
- package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
- package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
- package/dist/engine-CX8ReXkn.d.ts.map +0 -1
- package/dist/index-BxWvILU8.d.ts.map +0 -1
- package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
- package/dist/ledger-core-Cs9f7385.js.map +0 -1
- package/dist/llm-judge-v80Kmu9g.js.map +0 -1
- package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
- package/dist/outcome-store-ChBKlTd_.js +0 -75
- package/dist/outcome-store-ChBKlTd_.js.map +0 -1
- package/dist/promotion-policy-DWOm70gx.js.map +0 -1
- package/dist/report-command-DKlXfU5r.js.map +0 -1
- package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
- package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
- package/dist/run-record-CR63CpHK.js +0 -216
- package/dist/run-record-CR63CpHK.js.map +0 -1
- package/dist/sequential-B5gXgcyp.js.map +0 -1
- package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
- package/dist/terminal-record-Ce9_UjRz.js.map +0 -1
- package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
|
@@ -1,462 +1,11 @@
|
|
|
1
1
|
import { c as ValidationError, r as CaptureIntegrityError } from "../errors-DEE6u6ot.js";
|
|
2
2
|
import { n as bonferroni, r as holm, t as benjaminiHochberg } from "../multiplicity-DIWHvysC.js";
|
|
3
|
-
import { $ as paretoSignificanceGate, A as PairArmsOptions, B as PowerPreflight, C as hashJson, Dt as
|
|
4
|
-
import { c as BOOTSTRAP_GATE_MIN_N, d as PairedBootstrapResult, h as pairedBootstrap, u as PairedBootstrapOptions } from "../paired-promotion-decision-
|
|
5
|
-
import { _ as requiredSampleSize, a as ExperimentVerdict, d as inMemoryExperimentStore, f as mulberry32, g as requiredPairedSampleSize, h as pairedMde, i as ExperimentTracker, l as fileExperimentStore, m as mcnemarRequiredN, n as ExperimentRep, p as mcnemarPower, r as ExperimentStats, t as Experiment } from "../experiment-tracker-
|
|
3
|
+
import { $ as paretoSignificanceGate, $t as wilson, A as PairArmsOptions, At as FinalEvidenceReservation, B as PowerPreflight, Bt as eProcess, C as hashJson, Dt as FinalEvidenceMeasurement, E as verifyManifest, Et as FinalEvidenceLedger, Ft as summarizeEvaluationUnits, H as powerPreflight, It as EProcess, K as EvidenceVector, L as comparePairedArms, Lt as EProcessOptions, Mt as EvaluationClaim, N as PairedArmRow, Nt as EvaluationUnitSummary, O as MatchedPair, Ot as FinalEvidenceOutcome, Pt as defineEvaluationClaim, Q as paretoPolicy, R as pairArms, Rt as EProcessState, S as evaluateHypothesis, T as signManifest, Tt as FinalEvidenceError, Ut as ProportionInterval, V as PowerPreflightOptions, X as PromotionPolicy, Xt as pairedRiskDifferenceExact, Yt as pairedRiskDifference, Z as buildEvidenceVector, Zt as pairedRiskDifferenceScore, _ as sequentialPairedGate, _t as verifyEvidenceReceipt, at as createCampaignEvidenceReceipt, b as SignedManifest, c as pairHoldout, ct as EVIDENCE_RECEIPT_VERSION, d as SequentialDecision, dt as EvidenceBinding, f as SequentialObservation, ft as EvidenceReceipt, g as sequentialDecide, gt as isIndependentEvidence, ht as createEvidenceReceipt, i as PairedHoldout, it as CampaignEvidenceContext, j as PairArmsResult, jt as openFinalEvidenceLedger, kt as FinalEvidenceRecord, lt as EvidenceAuthority, m as SequentialPairedGateOptions, mt as INDEPENDENT_EVIDENCE_AUTHORITY_KINDS, n as HeldoutSignificance, ot as CreateEvidenceReceiptInput, p as SequentialPairedGate, pt as EvidenceReceiptVerification, qt as mcnemar, r as HeldoutSignificanceOptions, s as heldoutSignificance, st as EVIDENCE_AUTHORITY_KINDS, ut as EvidenceAuthorityKind, v as HypothesisManifest, w as manifestContentDigest, wt as FinalEvidenceConflictError, x as SignedManifestAlgo, y as HypothesisResult, z as pairRunRecords, zt as EProcessStep } from "../statistical-heldout-CpVd6FmY.js";
|
|
4
|
+
import { c as BOOTSTRAP_GATE_MIN_N, d as PairedBootstrapResult, h as pairedBootstrap, u as PairedBootstrapOptions } from "../paired-promotion-decision-DPsMQm-0.js";
|
|
5
|
+
import { _ as requiredSampleSize, a as ExperimentVerdict, d as inMemoryExperimentStore, f as mulberry32, g as requiredPairedSampleSize, h as pairedMde, i as ExperimentTracker, l as fileExperimentStore, m as mcnemarRequiredN, n as ExperimentRep, p as mcnemarPower, r as ExperimentStats, t as Experiment } from "../experiment-tracker-C7PfnF4b.js";
|
|
6
6
|
import { a as PairedEvalueStep, d as sequentialCrossingHorizon, i as PairedEvalueSequence, o as SequentialCrossingHorizon, r as PairedEvalueOptions, s as SequentialCrossingHorizonOptions, u as pairedEvalueSequence } from "../sequential-BhsrMupG.js";
|
|
7
|
+
import { A as classifyReissue, B as evaluateProvenanceGate, C as ReissuePolicy, D as UniformPassDecision, E as SetExpr, F as evaluateIdentityGate, G as runUniformPassBudget, H as projectNLadderBudget, I as evaluateOracleDeterminismGate, L as evaluatePopulationReproducibilityGate, M as computeInterval, N as evaluateCondition, O as UniformPassSchedule, P as evaluateHaltRule, R as evaluatePowerFloorGate, S as Predicate, T as SelectionRule, U as readField, V as executeDecisionRule, W as runSelectionRule, _ as IntervalSpec, a as ComputedInterval, b as NLadderProjection, c as DecisionOutcome, d as Estimand, f as EstimandResult, g as HaltRule, h as HaltOutcome, i as BudgetRule, j as computeEstimand, k as ValidityGate, l as DecisionRule, m as GateResult, n as AdmissionRule, o as Condition, p as EvidenceRecord, r as AdmissionStage, s as DecisionBranch, t as AdmissionPartition, u as DerivedQuantities, v as JsonValue, w as ReissueVerdict, x as Obligation, y as MatchedBudgetRule, z as evaluatePredicate } from "../ast-hI-vjW6J.js";
|
|
7
8
|
import { z } from "zod";
|
|
8
|
-
//#region src/experiment/ast.d.ts
|
|
9
|
-
/**
|
|
10
|
-
* The registered-rule AST: every rule an experiment registers is DATA.
|
|
11
|
-
*
|
|
12
|
-
* A closure cannot be canonicalized or hashed; a node tree can. `sealExperiment`
|
|
13
|
-
* hashes the whole tree, and every interpreter in this file takes only a node
|
|
14
|
-
* plus evidence records — no parameter for alpha, threshold, metric, or
|
|
15
|
-
* stopping rule exists on any executable surface. The registered object and
|
|
16
|
-
* the executed object are therefore the same object, and registered-vs-ran
|
|
17
|
-
* drift is unrepresentable rather than checked.
|
|
18
|
-
*
|
|
19
|
-
* Node families:
|
|
20
|
-
* Predicate closed-key comparisons — the only leaf
|
|
21
|
-
* AdmissionRule monotone funnel stages with registered waivers
|
|
22
|
-
* SelectionRule deterministic subsets over a closed field set
|
|
23
|
-
* Estimand what the experiment measures
|
|
24
|
-
* IntervalSpec how uncertainty is computed, seed included
|
|
25
|
-
* Condition decision guards over named derived quantities
|
|
26
|
-
* DecisionRule ordered verdict table, or a registered absence of one
|
|
27
|
-
* Obligation a control that must exist before a verdict class is read
|
|
28
|
-
* ValidityGate pre-spend design checks
|
|
29
|
-
* HaltRule gates as prerequisites — failure refuses the spend
|
|
30
|
-
* BudgetRule spend schedules with a named ledger
|
|
31
|
-
* MatchedBudgetRule arm budget matching as a refusal
|
|
32
|
-
* ReissuePolicy carrier faults are reissued; model outcomes stand
|
|
33
|
-
*/
|
|
34
|
-
/** JSON-serializable value — everything a sealed node may carry. */
|
|
35
|
-
type JsonValue = string | number | boolean | null | JsonValue[] | {
|
|
36
|
-
[k: string]: JsonValue;
|
|
37
|
-
};
|
|
38
|
-
/** Evidence row shape. Fields are addressed by dot-separated paths. */
|
|
39
|
-
type EvidenceRecord = Record<string, unknown>;
|
|
40
|
-
/**
|
|
41
|
-
* Closed-key comparison over a declared record schema. The only leaf node.
|
|
42
|
-
* `field` is a dot-separated path into an evidence record.
|
|
43
|
-
*/
|
|
44
|
-
type Predicate = {
|
|
45
|
-
kind: 'compare';
|
|
46
|
-
field: string;
|
|
47
|
-
op: 'eq' | 'ne' | 'lt' | 'lte' | 'gt' | 'gte';
|
|
48
|
-
value: JsonValue;
|
|
49
|
-
} | {
|
|
50
|
-
kind: 'in';
|
|
51
|
-
field: string;
|
|
52
|
-
values: JsonValue[];
|
|
53
|
-
} | {
|
|
54
|
-
kind: 'all';
|
|
55
|
-
of: Predicate[];
|
|
56
|
-
} | {
|
|
57
|
-
kind: 'any';
|
|
58
|
-
of: Predicate[];
|
|
59
|
-
} | {
|
|
60
|
-
kind: 'not';
|
|
61
|
-
of: Predicate;
|
|
62
|
-
};
|
|
63
|
-
/** Read a dot-separated field path. Missing segments yield `undefined`. */
|
|
64
|
-
declare function readField(record: EvidenceRecord, path: string): unknown;
|
|
65
|
-
/** Evaluate a predicate against one evidence record. */
|
|
66
|
-
declare function evaluatePredicate(predicate: Predicate, record: EvidenceRecord): boolean;
|
|
67
|
-
/**
|
|
68
|
-
* One monotone funnel stage. `waives` names substrate admission conditions
|
|
69
|
-
* deliberately NOT applied, so a waiver is registered, never implicit.
|
|
70
|
-
*/
|
|
71
|
-
interface AdmissionStage {
|
|
72
|
-
id: string;
|
|
73
|
-
keep: Predicate;
|
|
74
|
-
waives?: string[];
|
|
75
|
-
}
|
|
76
|
-
/** A partition is a set reported separately and never pooled. */
|
|
77
|
-
interface AdmissionPartition {
|
|
78
|
-
id: string;
|
|
79
|
-
/** Stage whose dropped rows this partition draws from. */
|
|
80
|
-
from: string;
|
|
81
|
-
keep: Predicate;
|
|
82
|
-
pooling: 'never';
|
|
83
|
-
}
|
|
84
|
-
/** Declarative row-admission funnel. Stages only remove rows. */
|
|
85
|
-
interface AdmissionRule {
|
|
86
|
-
population: string;
|
|
87
|
-
stages: AdmissionStage[];
|
|
88
|
-
partitions?: AdmissionPartition[];
|
|
89
|
-
}
|
|
90
|
-
/**
|
|
91
|
-
* Deterministic subset selection. `reads` is the closed field set the rule
|
|
92
|
-
* may touch — outcome-contaminated selection is unrepresentable because an
|
|
93
|
-
* outcome field is simply not in the list.
|
|
94
|
-
*/
|
|
95
|
-
type SelectionRule = {
|
|
96
|
-
kind: 'round-robin';
|
|
97
|
-
groupBy: string;
|
|
98
|
-
groupOrder: 'lex-asc';
|
|
99
|
-
withinOrder: {
|
|
100
|
-
field: string;
|
|
101
|
-
dir: 'asc' | 'desc';
|
|
102
|
-
};
|
|
103
|
-
take: number;
|
|
104
|
-
reads: string[];
|
|
105
|
-
} | {
|
|
106
|
-
kind: 'filter-of';
|
|
107
|
-
/** Name of the sealed subset or selection this rule filters. */
|
|
108
|
-
base: string;
|
|
109
|
-
keep: Predicate;
|
|
110
|
-
order: {
|
|
111
|
-
field: string;
|
|
112
|
-
dir: 'asc' | 'desc';
|
|
113
|
-
};
|
|
114
|
-
};
|
|
115
|
-
/**
|
|
116
|
-
* Execute a selection rule.
|
|
117
|
-
*
|
|
118
|
-
* Round-robin walks groups in lexicographic order and takes ids in
|
|
119
|
-
* within-group order until `take` ids are chosen. Filter-of keeps the ids of
|
|
120
|
-
* `bases[rule.base]` whose record satisfies the predicate, in the registered
|
|
121
|
-
* order. Ids absent from `records` are evaluated on their id alone (fields
|
|
122
|
-
* derived from the id via `idFields`), so a sealed base outlives its source
|
|
123
|
-
* records.
|
|
124
|
-
*/
|
|
125
|
-
declare function runSelectionRule(rule: SelectionRule, records: readonly EvidenceRecord[], options: {
|
|
126
|
-
/** Field carrying a row's identity. */
|
|
127
|
-
idField: string;
|
|
128
|
-
/** Sealed or previously-computed subsets, by name. */
|
|
129
|
-
bases?: Record<string, readonly string[]>;
|
|
130
|
-
/** Derive predicate-readable fields from a bare id when its record is absent. */
|
|
131
|
-
idFields?: (id: string) => EvidenceRecord;
|
|
132
|
-
}): string[];
|
|
133
|
-
/** A named set of row identities, built from arm rows and an event predicate. */
|
|
134
|
-
type SetExpr = {
|
|
135
|
-
kind: 'rows-where';
|
|
136
|
-
arm: string;
|
|
137
|
-
event: Predicate;
|
|
138
|
-
} | {
|
|
139
|
-
kind: 'intersect';
|
|
140
|
-
of: SetExpr[];
|
|
141
|
-
};
|
|
142
|
-
/**
|
|
143
|
-
* What the experiment measures. Every estimand names the fields it reads, so
|
|
144
|
-
* the sealed tree records the full data dependency of the number.
|
|
145
|
-
*/
|
|
146
|
-
type Estimand = {
|
|
147
|
-
kind: 'rate';
|
|
148
|
-
event: Predicate;
|
|
149
|
-
over: 'rollouts';
|
|
150
|
-
} | {
|
|
151
|
-
kind: 'rate-at-least-once';
|
|
152
|
-
event: Predicate;
|
|
153
|
-
groupBy: string;
|
|
154
|
-
} | {
|
|
155
|
-
kind: 'paired-mean-diff';
|
|
156
|
-
armField: string;
|
|
157
|
-
treatment: string;
|
|
158
|
-
control: string;
|
|
159
|
-
pairBy: string;
|
|
160
|
-
/**
|
|
161
|
-
* Field path to the per-row outcome. A boolean reads as 1 or 0, so a
|
|
162
|
-
* binary pass/fail outcome gives the risk difference directly.
|
|
163
|
-
*/
|
|
164
|
-
value: string;
|
|
165
|
-
/** A pair one arm did not answer contributes a difference of exactly zero. */
|
|
166
|
-
missing: 'zero-diff';
|
|
167
|
-
} | {
|
|
168
|
-
kind: 'set-ratio';
|
|
169
|
-
armField: string;
|
|
170
|
-
idField: string;
|
|
171
|
-
numerator: SetExpr;
|
|
172
|
-
denominator: SetExpr;
|
|
173
|
-
};
|
|
174
|
-
interface EstimandResult {
|
|
175
|
-
value: number;
|
|
176
|
-
numerator: number;
|
|
177
|
-
denominator: number;
|
|
178
|
-
}
|
|
179
|
-
/** Compute an estimand over evidence rows. Pure; reads only registered fields. */
|
|
180
|
-
declare function computeEstimand(estimand: Estimand, rows: readonly EvidenceRecord[]): EstimandResult;
|
|
181
|
-
/** How uncertainty is computed. The seed is part of the registration. */
|
|
182
|
-
type IntervalSpec = {
|
|
183
|
-
kind: 'cluster-bootstrap';
|
|
184
|
-
clusterBy: string;
|
|
185
|
-
resamples: number;
|
|
186
|
-
seed: number;
|
|
187
|
-
level: number;
|
|
188
|
-
method: 'percentile';
|
|
189
|
-
} | {
|
|
190
|
-
kind: 'clopper-pearson';
|
|
191
|
-
level: number;
|
|
192
|
-
};
|
|
193
|
-
interface ComputedInterval {
|
|
194
|
-
lower: number;
|
|
195
|
-
upper: number;
|
|
196
|
-
level: number;
|
|
197
|
-
}
|
|
198
|
-
/**
|
|
199
|
-
* Execute an interval spec.
|
|
200
|
-
*
|
|
201
|
-
* Cluster-bootstrap resamples whole clusters of the per-row `value` field and
|
|
202
|
-
* takes percentile bounds of the pooled mean. Clopper-Pearson computes the
|
|
203
|
-
* exact binomial interval and requires `successes`/`trials` evidence instead
|
|
204
|
-
* of rows.
|
|
205
|
-
*/
|
|
206
|
-
declare function computeInterval(spec: IntervalSpec, evidence: {
|
|
207
|
-
kind: 'rows';
|
|
208
|
-
rows: readonly EvidenceRecord[];
|
|
209
|
-
value: string;
|
|
210
|
-
} | {
|
|
211
|
-
kind: 'binomial';
|
|
212
|
-
successes: number;
|
|
213
|
-
trials: number;
|
|
214
|
-
}): ComputedInterval;
|
|
215
|
-
/** Decision guards read only named derived quantities — never raw rows. */
|
|
216
|
-
type Condition = {
|
|
217
|
-
kind: 'interval-excludes-zero';
|
|
218
|
-
interval: string;
|
|
219
|
-
sign: 'positive' | 'negative';
|
|
220
|
-
} | {
|
|
221
|
-
kind: 'interval-includes-zero';
|
|
222
|
-
interval: string;
|
|
223
|
-
} | {
|
|
224
|
-
kind: 'quantity-threshold';
|
|
225
|
-
quantity: string;
|
|
226
|
-
op: 'gte' | 'lte' | 'gt' | 'lt';
|
|
227
|
-
value: number;
|
|
228
|
-
} | {
|
|
229
|
-
kind: 'obligation-met';
|
|
230
|
-
obligation: string;
|
|
231
|
-
} | {
|
|
232
|
-
kind: 'all';
|
|
233
|
-
of: Condition[];
|
|
234
|
-
} | {
|
|
235
|
-
kind: 'any';
|
|
236
|
-
of: Condition[];
|
|
237
|
-
} | {
|
|
238
|
-
kind: 'not';
|
|
239
|
-
of: Condition;
|
|
240
|
-
};
|
|
241
|
-
/** The named quantities a decision rule may read. Nothing else reaches it. */
|
|
242
|
-
interface DerivedQuantities {
|
|
243
|
-
intervals: Record<string, {
|
|
244
|
-
lower: number;
|
|
245
|
-
upper: number;
|
|
246
|
-
}>;
|
|
247
|
-
quantities: Record<string, number>;
|
|
248
|
-
obligationsMet: Record<string, boolean>;
|
|
249
|
-
}
|
|
250
|
-
declare function evaluateCondition(condition: Condition, evidence: DerivedQuantities): boolean;
|
|
251
|
-
interface DecisionBranch {
|
|
252
|
-
when: Condition;
|
|
253
|
-
verdict: string;
|
|
254
|
-
report: string[];
|
|
255
|
-
}
|
|
256
|
-
/**
|
|
257
|
-
* Ordered decision table: the first branch whose condition holds fires, and a
|
|
258
|
-
* table no branch matches throws — a non-total registration is a defect, not
|
|
259
|
-
* an implicit verdict. `report-only` registers the ABSENCE of a verdict
|
|
260
|
-
* branch: the estimate and the per-row table are the finding, and the
|
|
261
|
-
* registered meaning prose rides as non-executable interpretation data.
|
|
262
|
-
*/
|
|
263
|
-
type DecisionRule = {
|
|
264
|
-
kind: 'table';
|
|
265
|
-
branches: DecisionBranch[];
|
|
266
|
-
} | {
|
|
267
|
-
kind: 'report-only';
|
|
268
|
-
estimands: string[];
|
|
269
|
-
intervals: string[];
|
|
270
|
-
perRow: string[];
|
|
271
|
-
interpretation?: {
|
|
272
|
-
onQualitative: string;
|
|
273
|
-
consequence: string;
|
|
274
|
-
}[];
|
|
275
|
-
};
|
|
276
|
-
interface DecisionOutcome {
|
|
277
|
-
verdict: string;
|
|
278
|
-
report: string[];
|
|
279
|
-
}
|
|
280
|
-
declare function executeDecisionRule(rule: DecisionRule, evidence: DerivedQuantities): DecisionOutcome;
|
|
281
|
-
/** A registered control that must exist before a class of verdicts is read. */
|
|
282
|
-
interface Obligation {
|
|
283
|
-
id: string;
|
|
284
|
-
appliesToVerdicts: string[];
|
|
285
|
-
control: string;
|
|
286
|
-
}
|
|
287
|
-
/** Pre-spend design checks. Each returns pass/fail plus its evidence. */
|
|
288
|
-
type ValidityGate = {
|
|
289
|
-
kind: 'oracle-determinism';
|
|
290
|
-
unit: 'suite' | 'assertion';
|
|
291
|
-
replicates: number;
|
|
292
|
-
maxFlipRate: number;
|
|
293
|
-
} | {
|
|
294
|
-
kind: 'population-reproducibility';
|
|
295
|
-
joinOn: string;
|
|
296
|
-
compare: string[];
|
|
297
|
-
maxChangedRows: number;
|
|
298
|
-
} | {
|
|
299
|
-
kind: 'provenance-assertion';
|
|
300
|
-
claim: Predicate;
|
|
301
|
-
} | {
|
|
302
|
-
kind: 'power-floor';
|
|
303
|
-
target: number;
|
|
304
|
-
effectGrid: number[];
|
|
305
|
-
sim: {
|
|
306
|
-
trials: number;
|
|
307
|
-
resamples: number;
|
|
308
|
-
seed: number;
|
|
309
|
-
};
|
|
310
|
-
} | {
|
|
311
|
-
kind: 'identity';
|
|
312
|
-
field: 'served-model';
|
|
313
|
-
op: 'basename-eq';
|
|
314
|
-
onFail: 'abort';
|
|
315
|
-
};
|
|
316
|
-
interface GateResult {
|
|
317
|
-
id: string;
|
|
318
|
-
passed: boolean;
|
|
319
|
-
evidence: JsonValue;
|
|
320
|
-
}
|
|
321
|
-
/**
|
|
322
|
-
* Replicate-flip counting over graded states. A state whose replicates split
|
|
323
|
-
* between pass and fail is flipping; its flip rate is the minority share.
|
|
324
|
-
*/
|
|
325
|
-
declare function evaluateOracleDeterminismGate(id: string, gate: Extract<ValidityGate, {
|
|
326
|
-
kind: 'oracle-determinism';
|
|
327
|
-
}>, repsByState: Record<string, readonly boolean[]>): GateResult;
|
|
328
|
-
/**
|
|
329
|
-
* Join two population snapshots on `joinOn` and compare the registered fields.
|
|
330
|
-
* Only rows present in both snapshots are compared; a presence change is a
|
|
331
|
-
* different failure and needs its own gate.
|
|
332
|
-
*/
|
|
333
|
-
declare function evaluatePopulationReproducibilityGate(id: string, gate: Extract<ValidityGate, {
|
|
334
|
-
kind: 'population-reproducibility';
|
|
335
|
-
}>, populations: {
|
|
336
|
-
left: readonly EvidenceRecord[];
|
|
337
|
-
right: readonly EvidenceRecord[];
|
|
338
|
-
}): GateResult;
|
|
339
|
-
/** The registered claim about provenance must hold on the provenance record. */
|
|
340
|
-
declare function evaluateProvenanceGate(id: string, gate: Extract<ValidityGate, {
|
|
341
|
-
kind: 'provenance-assertion';
|
|
342
|
-
}>, provenance: EvidenceRecord): GateResult;
|
|
343
|
-
/** Final path segment equality between the pinned and the served identity. */
|
|
344
|
-
declare function evaluateIdentityGate(id: string, _gate: Extract<ValidityGate, {
|
|
345
|
-
kind: 'identity';
|
|
346
|
-
}>, identities: {
|
|
347
|
-
pinned: string;
|
|
348
|
-
served: string;
|
|
349
|
-
}): GateResult;
|
|
350
|
-
/**
|
|
351
|
-
* The design's power curve must reach the registered target at some grid
|
|
352
|
-
* effect. The curve must cover the registered effect grid exactly — a curve
|
|
353
|
-
* computed on a different grid is different evidence and is refused.
|
|
354
|
-
*/
|
|
355
|
-
declare function evaluatePowerFloorGate(id: string, gate: Extract<ValidityGate, {
|
|
356
|
-
kind: 'power-floor';
|
|
357
|
-
}>, curve: readonly {
|
|
358
|
-
effect: number;
|
|
359
|
-
power: number;
|
|
360
|
-
}[]): GateResult;
|
|
361
|
-
/** Checks as prerequisites: any named gate failing refuses the spend. */
|
|
362
|
-
interface HaltRule {
|
|
363
|
-
when: {
|
|
364
|
-
kind: 'any-gate-failed';
|
|
365
|
-
gates: string[];
|
|
366
|
-
};
|
|
367
|
-
action: 'refuse-spend';
|
|
368
|
-
report: 'settling-n';
|
|
369
|
-
}
|
|
370
|
-
interface HaltOutcome {
|
|
371
|
-
fired: boolean;
|
|
372
|
-
action: 'refuse-spend' | null;
|
|
373
|
-
failedGates: string[];
|
|
374
|
-
}
|
|
375
|
-
declare function evaluateHaltRule(halt: HaltRule, gates: readonly GateResult[]): HaltOutcome;
|
|
376
|
-
/**
|
|
377
|
-
* Spend schedules as data. `ledger` names every entry the ceiling counts, so
|
|
378
|
-
* what the gate reads is registered, not improvised at gate time.
|
|
379
|
-
*/
|
|
380
|
-
type BudgetRule = {
|
|
381
|
-
kind: 'uniform-pass';
|
|
382
|
-
ceilingUsd: number;
|
|
383
|
-
maxPasses: number;
|
|
384
|
-
projection: 'last-pass-cost';
|
|
385
|
-
costSource: 'priced-per-call';
|
|
386
|
-
ledger: {
|
|
387
|
-
id: string;
|
|
388
|
-
usd: number;
|
|
389
|
-
}[];
|
|
390
|
-
partialPass: 'report-never-lift';
|
|
391
|
-
} | {
|
|
392
|
-
kind: 'n-ladder';
|
|
393
|
-
steps: number[];
|
|
394
|
-
ceilingUsd: number;
|
|
395
|
-
projection: 'unit-cost-times-rows-times-n';
|
|
396
|
-
onExhaust: 'refuse-report-projection';
|
|
397
|
-
};
|
|
398
|
-
interface UniformPassDecision {
|
|
399
|
-
pass: number;
|
|
400
|
-
cumulativeBefore: number;
|
|
401
|
-
projected: number;
|
|
402
|
-
go: boolean;
|
|
403
|
-
}
|
|
404
|
-
interface UniformPassSchedule {
|
|
405
|
-
decisions: UniformPassDecision[];
|
|
406
|
-
uniformN: number;
|
|
407
|
-
}
|
|
408
|
-
/**
|
|
409
|
-
* Execute the uniform-pass schedule against measured pass costs. Pass 1 always
|
|
410
|
-
* runs; each later pass runs only when the cumulative spend plus the last
|
|
411
|
-
* measured pass cost stays at or under the registered ceiling. The registered
|
|
412
|
-
* ledger is the pre-spend the ceiling counts.
|
|
413
|
-
*/
|
|
414
|
-
declare function runUniformPassBudget(rule: Extract<BudgetRule, {
|
|
415
|
-
kind: 'uniform-pass';
|
|
416
|
-
}>, measuredPassCosts: readonly number[]): UniformPassSchedule;
|
|
417
|
-
interface NLadderProjection {
|
|
418
|
-
chosenN: number | null;
|
|
419
|
-
projections: {
|
|
420
|
-
n: number;
|
|
421
|
-
projectedUsd: number;
|
|
422
|
-
affordable: boolean;
|
|
423
|
-
}[];
|
|
424
|
-
refusal: null | {
|
|
425
|
-
onExhaust: 'refuse-report-projection';
|
|
426
|
-
reason: string;
|
|
427
|
-
};
|
|
428
|
-
}
|
|
429
|
-
/**
|
|
430
|
-
* Walk the registered n-ladder and pick the first affordable step. When no
|
|
431
|
-
* step fits the ceiling, the rule refuses and reports the projection instead
|
|
432
|
-
* of shrinking the row set — "never subset rows" is the registered invariant.
|
|
433
|
-
*/
|
|
434
|
-
declare function projectNLadderBudget(rule: Extract<BudgetRule, {
|
|
435
|
-
kind: 'n-ladder';
|
|
436
|
-
}>, measured: {
|
|
437
|
-
unitCostUsd: number;
|
|
438
|
-
rows: number;
|
|
439
|
-
}): NLadderProjection;
|
|
440
|
-
/** Arm budget matching as a refusal, not prose. Verified by `verifyMatchedBudgets`. */
|
|
441
|
-
interface MatchedBudgetRule {
|
|
442
|
-
measure: 'realized-tokens';
|
|
443
|
-
tolerance: number;
|
|
444
|
-
onFail: 'refuse-contrast';
|
|
445
|
-
}
|
|
446
|
-
/** Carrier faults are reissued; model outcomes stand. Closed enumeration. */
|
|
447
|
-
interface ReissuePolicy {
|
|
448
|
-
carrierEvents: ('http-status' | 'transport-error' | 'deadline' | 'empty-content')[];
|
|
449
|
-
modelOutcomesStand: true;
|
|
450
|
-
maxIssues: number;
|
|
451
|
-
}
|
|
452
|
-
type ReissueVerdict = 'reissue' | 'stands' | 'exhausted';
|
|
453
|
-
/**
|
|
454
|
-
* Classify one rollout event under the registered reissue policy. A carrier
|
|
455
|
-
* event within the issue budget is reissued; a model outcome always stands;
|
|
456
|
-
* a carrier event past `maxIssues` is exhausted and reported, never retried.
|
|
457
|
-
*/
|
|
458
|
-
declare function classifyReissue(policy: ReissuePolicy, event: string, issuesSoFar: number): ReissueVerdict;
|
|
459
|
-
//#endregion
|
|
460
9
|
//#region src/experiment/budget.d.ts
|
|
461
10
|
/** Arms whose realized budgets diverge past the registered tolerance. */
|
|
462
11
|
declare class MatchedBudgetError extends CaptureIntegrityError {}
|
|
@@ -607,6 +156,8 @@ interface SeedDerivation {
|
|
|
607
156
|
}
|
|
608
157
|
interface ExperimentSpec {
|
|
609
158
|
id: string;
|
|
159
|
+
/** Population and independent observation unit for the intended use of this result. */
|
|
160
|
+
claim?: EvaluationClaim;
|
|
610
161
|
/** Human prose for the audit trail — never executable. */
|
|
611
162
|
hypothesis?: string;
|
|
612
163
|
arms: ArmSpec[];
|
|
@@ -650,21 +201,13 @@ interface SealAmendment {
|
|
|
650
201
|
/** Digest of the spec after this amendment. */
|
|
651
202
|
digest: string;
|
|
652
203
|
}
|
|
653
|
-
/**
|
|
654
|
-
|
|
655
|
-
* serialized spec and differ only in the serialization: `'sha256-rfc8785'` is
|
|
656
|
-
* RFC 8785 canonical JSON, `'sha256-content'` is key-sorted `JSON.stringify`.
|
|
657
|
-
*/
|
|
658
|
-
type SealAlgo = 'sha256-content' | 'sha256-rfc8785';
|
|
204
|
+
/** SHA-256 over the RFC 8785 canonical JSON encoding of the experiment. */
|
|
205
|
+
type SealAlgo = 'sha256-rfc8785';
|
|
659
206
|
interface SealedExperiment {
|
|
660
207
|
spec: ExperimentSpec;
|
|
661
208
|
/** sha256 over the serialized spec, under the scheme `algo` names. */
|
|
662
209
|
digest: string;
|
|
663
|
-
/**
|
|
664
|
-
* Digest scheme of `digest`. `'sha256-rfc8785'` is what {@link sealExperiment}
|
|
665
|
-
* emits; `'sha256-content'` is read-only, carried by seals from an earlier
|
|
666
|
-
* release, and still verifies.
|
|
667
|
-
*/
|
|
210
|
+
/** Required digest scheme. Unsupported or missing schemes cannot execute. */
|
|
668
211
|
algo: SealAlgo;
|
|
669
212
|
sealedAt: string;
|
|
670
213
|
/** Digest of the original registration, before any amendment. */
|
|
@@ -686,8 +229,7 @@ declare function amendExperiment(sealed: SealedExperiment, amendment: {
|
|
|
686
229
|
blind: string[];
|
|
687
230
|
at?: string;
|
|
688
231
|
}): Promise<SealedExperiment>;
|
|
689
|
-
/** True when the
|
|
690
|
-
* scheme the seal declares. */
|
|
232
|
+
/** True when the supported seal matches its canonical spec. */
|
|
691
233
|
declare function verifySealedExperiment(sealed: SealedExperiment): Promise<boolean>;
|
|
692
234
|
/** Evidence a gate executor consumes, discriminated to match the gate kind. */
|
|
693
235
|
type GateEvidence = {
|
|
@@ -717,6 +259,8 @@ type GateEvidence = {
|
|
|
717
259
|
*/
|
|
718
260
|
interface RegisteredExperiment {
|
|
719
261
|
readonly sealed: SealedExperiment;
|
|
262
|
+
/** Count independent units separately from repeated observations under the sealed claim. */
|
|
263
|
+
units(records: readonly EvidenceRecord[]): EvaluationUnitSummary;
|
|
720
264
|
/** Execute the registered decision rule on derived quantities. */
|
|
721
265
|
decide(evidence: DerivedQuantities): DecisionOutcome;
|
|
722
266
|
/** Run the registered admission funnel over evidence rows. */
|
|
@@ -744,12 +288,14 @@ interface RegisteredExperiment {
|
|
|
744
288
|
interval(name: string, evidence: {
|
|
745
289
|
kind: 'rows';
|
|
746
290
|
rows: readonly EvidenceRecord[];
|
|
747
|
-
value: string;
|
|
748
291
|
} | {
|
|
749
292
|
kind: 'binomial';
|
|
750
293
|
successes: number;
|
|
751
294
|
trials: number;
|
|
752
|
-
|
|
295
|
+
unitIds?: readonly string[];
|
|
296
|
+
}): ComputedInterval & {
|
|
297
|
+
units?: EvaluationUnitSummary;
|
|
298
|
+
};
|
|
753
299
|
}
|
|
754
300
|
/**
|
|
755
301
|
* Verify the seal and return executors bound to it. This is the module's only
|
|
@@ -831,8 +377,10 @@ declare class DesignRefusalError extends ValidationError {}
|
|
|
831
377
|
interface ClusteredPowerOptions {
|
|
832
378
|
/** Rows per independent cluster, e.g. [6, 3, 3, 2]. */
|
|
833
379
|
clusterSizes: number[];
|
|
834
|
-
/**
|
|
380
|
+
/** P(win) − P(loss) in signal clusters; noisy clusters retain zero expected contrast. */
|
|
835
381
|
effects: number[];
|
|
382
|
+
/** Minimum worthwhile effect under the declared outcome model; must occur in effects. */
|
|
383
|
+
minimumEffect: number;
|
|
836
384
|
/** Deterministic seed for outcome draws and bootstrap resampling. */
|
|
837
385
|
seed: number;
|
|
838
386
|
/** Simulated experiments per effect. Default 2000. */
|
|
@@ -843,11 +391,11 @@ interface ClusteredPowerOptions {
|
|
|
843
391
|
confidence?: number;
|
|
844
392
|
/** Sign-flip alpha the closed-form floor is checked against. Default 0.05. */
|
|
845
393
|
alpha?: number;
|
|
846
|
-
/** Power
|
|
394
|
+
/** Power required at minimumEffect. Default 0.8. */
|
|
847
395
|
targetPower?: number;
|
|
848
|
-
/**
|
|
396
|
+
/** P(row favors treatment) under the zero-effect model. Must equal baseLossRate. Default 0.10. */
|
|
849
397
|
baseWinRate?: number;
|
|
850
|
-
/**
|
|
398
|
+
/** P(row favors control) under the zero-effect model. Must equal baseWinRate. Default 0.10. */
|
|
851
399
|
baseLossRate?: number;
|
|
852
400
|
/**
|
|
853
401
|
* Clusters whose rows carry outcome noise instead of signal: each row wins
|
|
@@ -888,22 +436,24 @@ interface ClusteredPowerResult {
|
|
|
888
436
|
seed: number;
|
|
889
437
|
confidence: number;
|
|
890
438
|
targetPower: number;
|
|
439
|
+
minimumEffect: number;
|
|
440
|
+
powerAtMinimumEffect: number;
|
|
891
441
|
curve: ClusteredPowerPoint[];
|
|
892
442
|
/** Maximum simulated power across the effect grid. */
|
|
893
443
|
maxPower: number;
|
|
894
444
|
signFlipFloor: SignFlipFloor;
|
|
895
|
-
/** True
|
|
445
|
+
/** True when the sign-flip floor certifies and power at minimumEffect reaches target. */
|
|
896
446
|
adequate: boolean;
|
|
897
447
|
/** Populated exactly when `adequate` is false. The refusal lives in the artifact. */
|
|
898
448
|
refusal: ClusteredPowerRefusal | null;
|
|
899
449
|
}
|
|
900
450
|
/**
|
|
901
451
|
* Simulate the power of a whole-cluster percentile-bootstrap design and refuse
|
|
902
|
-
* a structure that cannot reach
|
|
452
|
+
* a structure that cannot reach target power at the declared minimum effect.
|
|
903
453
|
*/
|
|
904
454
|
declare function clusteredPower(options: ClusteredPowerOptions): ClusteredPowerResult;
|
|
905
455
|
/** Throw the refusal for callers that want configuration-time failure. */
|
|
906
456
|
declare function assertDesignAdequate(result: ClusteredPowerResult): void;
|
|
907
457
|
//#endregion
|
|
908
|
-
export { type AdmissionExecution, type AdmissionPartition, type AdmissionRule, type AdmissionStage, type ArmRealizedBudget, type ArmSpec, BOOTSTRAP_GATE_MIN_N, type BudgetRule, type CampaignEvidenceContext, type ClusteredPowerOptions, type ClusteredPowerPoint, type ClusteredPowerRefusal, type ClusteredPowerResult, type ComputedInterval, type Condition, type CreateEvidenceReceiptInput, type DecisionBranch, type DecisionOutcome, type DecisionRule, type DerivedQuantities, DesignRefusalError, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, EVIDENCE_AUTHORITY_KINDS, EVIDENCE_RECEIPT_VERSION, EVIDENCE_STATES, type Estimand, type EstimandResult, type EvidenceAuthority, type EvidenceAuthorityKind, type EvidenceBinding, type EvidenceDenominator, type EvidenceReceipt, type EvidenceReceiptVerification, type EvidenceRecord, type EvidenceRegistryRecord, type EvidenceState, type EvidenceVector, type Experiment, type ExperimentFunnel, type ExperimentRep, type ExperimentSpec, type ExperimentStats, ExperimentTracker, type ExperimentVerdict, FunnelIntegrityError, type FunnelPartitionCount, type FunnelStageCount, type FunnelStageInput, type GateEvidence, type GateResult, type HaltOutcome, type HaltRule, type HeldoutSignificance, type HeldoutSignificanceOptions, type HypothesisManifest, type HypothesisResult, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS, type IntervalSpec, type JsonValue, MatchedBudgetError, type MatchedBudgetRule, type MatchedBudgetVerdict, type MatchedPair, type NLadderProjection, type Obligation, type OutcomeSpec, type PairArmsOptions, type PairArmsResult, type PairedArmRow, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedHoldout, type PowerPreflight, type PowerPreflightOptions, type Predicate, type PromotionPolicy, type ProportionInterval, type RegisteredExperiment, type ReissuePolicy, type ReissueVerdict, type SealAmendment, SealIntegrityError, type SealedExperiment, type SeedDerivation, type SelectionRule, type SequentialCrossingHorizon, type SequentialCrossingHorizonOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SetExpr, type SignFlipFloor, type SignedManifest, type SignedManifestAlgo, type UniformPassDecision, type UniformPassSchedule, type ValidityGate, amendExperiment, assertDesignAdequate, assertFunnelReconciles, assertMatchedBudgets, benjaminiHochberg, bonferroni, buildEvidenceVector, buildFunnel, classifyReissue, clusteredPower, comparePairedArms, composeFunnels, computeEstimand, computeInterval, createCampaignEvidenceReceipt, createEvidenceReceipt, defineExperiment, eProcess, evaluateCondition, evaluateHaltRule, evaluateHypothesis, evaluateIdentityGate, evaluateOracleDeterminismGate, evaluatePopulationReproducibilityGate, evaluatePowerFloorGate, evaluatePredicate, evaluateProvenanceGate, evidenceRegistryRecordSchema, executeAdmissionRule, executeDecisionRule, fileExperimentStore, hashJson, heldoutSignificance, holm, inMemoryExperimentStore, isIndependentEvidence, manifestContentDigest, mcnemar, mcnemarPower, mcnemarRequiredN, mulberry32, openSealedExperiment, pairArms, pairHoldout, pairRunRecords, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, paretoPolicy, paretoSignificanceGate, parseEvidenceRegistryRecord, powerPreflight, projectNLadderBudget, readField, renderEvidenceIndex, renderFunnelTable, requiredPairedSampleSize, requiredSampleSize, runSelectionRule, runUniformPassBudget, sealExperiment, sequentialCrossingHorizon, sequentialDecide, sequentialPairedGate, signManifest, validateEvidenceRegistry, verifyEvidenceReceipt, verifyManifest, verifyMatchedBudgets, verifySealedExperiment, wilson };
|
|
458
|
+
export { type AdmissionExecution, type AdmissionPartition, type AdmissionRule, type AdmissionStage, type ArmRealizedBudget, type ArmSpec, BOOTSTRAP_GATE_MIN_N, type BudgetRule, type CampaignEvidenceContext, type ClusteredPowerOptions, type ClusteredPowerPoint, type ClusteredPowerRefusal, type ClusteredPowerResult, type ComputedInterval, type Condition, type CreateEvidenceReceiptInput, type DecisionBranch, type DecisionOutcome, type DecisionRule, type DerivedQuantities, DesignRefusalError, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, EVIDENCE_AUTHORITY_KINDS, EVIDENCE_RECEIPT_VERSION, EVIDENCE_STATES, type Estimand, type EstimandResult, type EvaluationClaim, type EvaluationUnitSummary, type EvidenceAuthority, type EvidenceAuthorityKind, type EvidenceBinding, type EvidenceDenominator, type EvidenceReceipt, type EvidenceReceiptVerification, type EvidenceRecord, type EvidenceRegistryRecord, type EvidenceState, type EvidenceVector, type Experiment, type ExperimentFunnel, type ExperimentRep, type ExperimentSpec, type ExperimentStats, ExperimentTracker, type ExperimentVerdict, FinalEvidenceConflictError, FinalEvidenceError, type FinalEvidenceLedger, type FinalEvidenceMeasurement, type FinalEvidenceOutcome, type FinalEvidenceRecord, type FinalEvidenceReservation, FunnelIntegrityError, type FunnelPartitionCount, type FunnelStageCount, type FunnelStageInput, type GateEvidence, type GateResult, type HaltOutcome, type HaltRule, type HeldoutSignificance, type HeldoutSignificanceOptions, type HypothesisManifest, type HypothesisResult, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS, type IntervalSpec, type JsonValue, MatchedBudgetError, type MatchedBudgetRule, type MatchedBudgetVerdict, type MatchedPair, type NLadderProjection, type Obligation, type OutcomeSpec, type PairArmsOptions, type PairArmsResult, type PairedArmRow, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedHoldout, type PowerPreflight, type PowerPreflightOptions, type Predicate, type PromotionPolicy, type ProportionInterval, type RegisteredExperiment, type ReissuePolicy, type ReissueVerdict, type SealAmendment, SealIntegrityError, type SealedExperiment, type SeedDerivation, type SelectionRule, type SequentialCrossingHorizon, type SequentialCrossingHorizonOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SetExpr, type SignFlipFloor, type SignedManifest, type SignedManifestAlgo, type UniformPassDecision, type UniformPassSchedule, type ValidityGate, amendExperiment, assertDesignAdequate, assertFunnelReconciles, assertMatchedBudgets, benjaminiHochberg, bonferroni, buildEvidenceVector, buildFunnel, classifyReissue, clusteredPower, comparePairedArms, composeFunnels, computeEstimand, computeInterval, createCampaignEvidenceReceipt, createEvidenceReceipt, defineEvaluationClaim, defineExperiment, eProcess, evaluateCondition, evaluateHaltRule, evaluateHypothesis, evaluateIdentityGate, evaluateOracleDeterminismGate, evaluatePopulationReproducibilityGate, evaluatePowerFloorGate, evaluatePredicate, evaluateProvenanceGate, evidenceRegistryRecordSchema, executeAdmissionRule, executeDecisionRule, fileExperimentStore, hashJson, heldoutSignificance, holm, inMemoryExperimentStore, isIndependentEvidence, manifestContentDigest, mcnemar, mcnemarPower, mcnemarRequiredN, mulberry32, openFinalEvidenceLedger, openSealedExperiment, pairArms, pairHoldout, pairRunRecords, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, paretoPolicy, paretoSignificanceGate, parseEvidenceRegistryRecord, powerPreflight, projectNLadderBudget, readField, renderEvidenceIndex, renderFunnelTable, requiredPairedSampleSize, requiredSampleSize, runSelectionRule, runUniformPassBudget, sealExperiment, sequentialCrossingHorizon, sequentialDecide, sequentialPairedGate, signManifest, summarizeEvaluationUnits, validateEvidenceRegistry, verifyEvidenceReceipt, verifyManifest, verifyMatchedBudgets, verifySealedExperiment, wilson };
|
|
909
459
|
//# sourceMappingURL=index.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/experiment/
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/experiment/budget.ts","../../src/experiment/funnel.ts","../../src/experiment/define.ts","../../src/experiment/evidence-record.ts","../../src/experiment/power.ts"],"mappings":";;;;;;;;;;cAaa,2BAA2B;UAEvB;EACf;;EAEA;;UAGe;EACf,MAAM;EACN,MAAM;;EAEN;;EAEA;EACA;;EAEA;IAAkB;IAA2B;;;;;;;;iBAQ/B,qBACd,MAAM,mBACN,eAAe,sBACd;;iBAsCa,qBACd,MAAM,mBACN,eAAe,sBACd;;;;cC/DU,6BAA6B;UAEzB;EACf;EACA;EACA;EACA;;EAEA,aAAa;;EAEb;;UAGe;EACf;;EAEA;EACA;;EAEA;;UAGe;EACf;EACA;EACA,QAAQ;EACR;EACA,YAAY;;UAGG;EACf;;EAEA;EACA,aAAa;EACb;;;;;;;;;;;iBAYc,YAAY;EAC1B;EACA;EACA,QAAQ;EACR;IAAe;IAAY;IAAc;;IACvC;;iBAsEY,uBAAuB,QAAQ;UAwB9B;EACf,QAAQ;EACR,WAAW;;EAEX,eAAe,eAAe;;;;;;;;iBAShB,qBACd,MAAM,eACN,kBAAkB,mBACjB;;;;;;iBA8Ca,eACd,OAAO,kBACP,QAAQ,mBACP;;;;;;iBA4Ba,kBAAkB,QAAQ;;;;cC/L7B,2BAA2B;UAIvB;EACf;EACA;;EAEA;;EAEA;;EAEA,OAAO,eAAe;;KAGZ;EAEN;;EAEA;EACA;;EAEA;;EAEA;;EAGA;EACA;EACA;EACA;;;;;;UAOW;EACf;;UAGe;EACf;;EAEA,QAAQ;;EAER;EACA,MAAM;EACN,SAAS;EACT,YAAY;;EAEZ,aAAa,eAAe;;;;;EAK5B,gBAAgB;EAChB,YAAY,eAAe;EAC3B,YAAY,eAAe;EAC3B,UAAU;EACV,cAAc;;EAEd,QAAQ,eAAe;EACvB,OAAO;EACP,SAAS;EACT,gBAAgB;EAChB,UAAU;EACV,iBAAiB;EACjB;;;;;;;;;;iBAwCc,iBAAiB,MAAM,iBAAiB;UA0IvC;;EAEf;EACA;;EAEA;;EAEA;;;KAIU;UAEK;EACf,MAAM;;EAEN;;EAEA,MAAM;EACN;;EAEA;EACA,YAAY;;;iBAIQ,eACpB,MAAM,gBACN;EAAW;IACV,QAAQ;;;;;;iBAkBW,gBACpB,QAAQ,kBACR;EAAa,MAAM;EAAgB;EAAgB;EAAiB;IACnE,QAAQ;;iBAyBW,uBAAuB,QAAQ,mBAAmB;;KAwB5D;EACN;EAA4B,aAAa;;EAEzC;EACA,eAAe;EACf,gBAAgB;;EAEhB;EAA8B,YAAY;;EAC1C;EAAkB;EAAgB;;EAClC;EAAqB;IAAkB;IAAgB;;;;;;;UAM5C;WACN,QAAQ;;EAEjB,MAAM,kBAAkB,mBAAmB;;EAE3C,OAAO,UAAU,oBAAoB;;EAErC,MAAM,kBAAkB,mBAAmB;;EAE3C,OAAO,cAAc,kBAAkB,kBAAkB;IAAW;;;EAEpE,KAAK,cAAc,UAAU,eAAe;;EAE5C,KAAK,gBAAgB,eAAe;;EAEpC,qBAAqB,uCAAuC;;EAE5D,qBAAqB;IAAY;IAAqB;MAAiB;;EAEvE,eAAe,eAAe,sBAAsB;;EAEpD,SAAS,cAAc,eAAe,mBAAmB;;EAEzD,SACE,cACA;IACM;IAAc,eAAe;;IAC7B;IAAkB;IAAmB;IAAgB;MAC1D;IAAqB,QAAQ;;;;;;;;iBAQZ,qBACpB,QAAQ,mBACP,QAAQ;;;;;;;;;;;;;;;cC5aE;KAQD,wBAAwB;;cAO9B,2BAAyB,EAAA;;;;GAM7B,EAAA,KAAA;KAEU,sBAAsB,EAAE,aAAa;cAEpC,8BAA4B,EAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAsCvC,EAAA,KAAA;KAEU,yBAAyB,EAAE,aAAa;;iBAGpC,4BAA4B,eAAe;;;;;;iBAW3C,yBAAyB,2BAA2B;;;;;;iBA4CpD,oBAAoB;;;;cCpIvB,2BAA2B;UAEvB;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;EAMA;IAAkB;IAAe;;;UAGlB;EACf;EACA;EACA;;UAGe;;EAEf;;EAEA;EACA;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,OAAO;;EAEP;EACA,eAAe;;EAEf;;EAEA,SAAS;;;;;;iBAOK,eAAe,SAAS,wBAAwB;;iBAgKhD,qBAAqB,QAAQ"}
|