@tangle-network/agent-eval 0.144.6 → 0.144.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +470 -88
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BKENp2s5.js} +662 -605
- package/dist/benchmark-command-BKENp2s5.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BEOkuvIg.js → benchmarks-BlPmjd88.js} +4 -4
- package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-BlPmjd88.js.map} +1 -1
- package/dist/campaign/index.d.ts +7 -5
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-CXsdyym7.js → campaign--HVSuvV0.js} +17 -301
- package/dist/campaign--HVSuvV0.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
- package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
- package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -8
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
- package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
- package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +214 -404
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +233 -649
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
- package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
- package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
- package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +2 -2
- package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
- package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +17 -7
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
- package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
- package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
- package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CkaI2ly4.js} +19 -654
- package/dist/skillopt-optimization-method-CkaI2ly4.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
- package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
- package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +2 -2
- package/dist/wire/index.js +1 -1
- package/docs/campaign-proposers.md +5 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +19 -2
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
- package/dist/benchmark-command-CQd78YHt.js.map +0 -1
- package/dist/campaign-CXsdyym7.js.map +0 -1
- package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
- package/dist/index-4XwggC10.d.ts.map +0 -1
- package/dist/index-BIL5vxxt.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-Dv5BiKLE.js.map +0 -1
- package/dist/replay-Krvb114g.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
- package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
- package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
- package/dist/types-XMVEdrE_.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -0,0 +1,802 @@
|
|
|
1
|
+
import { l as ValidationError, r as CaptureIntegrityError } from "../errors-CKPfb2aH.js";
|
|
2
|
+
import { B as eProcess, J as mcnemarRequiredN, K as mcnemar, N as benjaminiHochberg, P as bonferroni, Q as pairedBootstrap, S as ProportionInterval, V as holm, Y as mulberry32, _t as wilson, d as EProcessState, dt as requiredPairedSampleSize, f as EProcessStep, ft as requiredSampleSize, it as pairedRiskDifferenceScore, l as EProcess, nt as pairedRiskDifference, q as mcnemarPower, rt as pairedRiskDifferenceExact, t as BOOTSTRAP_GATE_MIN_N, tt as pairedMde, u as EProcessOptions, v as PairedBootstrapOptions, y as PairedBootstrapResult } from "../statistics-D6Uebe_4.js";
|
|
3
|
+
import { A as PairArmsResult, C as hashJson, D as MatchedPair, I as comparePairedArms, L as pairArms, M as PairedArmRow, R as pairRunRecords, S as evaluateHypothesis, T as verifyManifest, _ as HypothesisManifest, c as pairHoldout, d as SequentialDecision, f as SequentialObservation, g as sequentialPairedGate, h as sequentialDecide, i as PairedHoldout, k as PairArmsOptions, m as SequentialPairedGateOptions, n as HeldoutSignificance, p as SequentialPairedGate, r as HeldoutSignificanceOptions, s as heldoutSignificance, v as HypothesisResult, w as signManifest, x as canonicalize, y as SignedManifest } from "../statistical-heldout-Dn9ruizm.js";
|
|
4
|
+
import { S as powerPreflight, b as PowerPreflight, c as PromotionPolicy, d as paretoSignificanceGate, i as EvidenceVector, l as buildEvidenceVector, u as paretoPolicy, x as PowerPreflightOptions } from "../promotion-policy-ChWhTDBH.js";
|
|
5
|
+
import { C as ClusteredPairedBinaryOptions, E as clusteredPairedBinary, S as ClusteredMatchedPair, T as ClusteredPairedBinaryStatistics, _ as inMemoryExperimentStore, a as ExperimentStats, b as ClusterSignFlipResult, i as ExperimentRep, l as ExperimentVerdict, m as fileExperimentStore, n as Experiment, s as ExperimentTracker, v as ClusterBootstrapInterval, w as ClusteredPairedBinaryResult, x as ClusteredBinaryCluster } from "../experiment-tracker-IMntXr6J.js";
|
|
6
|
+
import { a as PairedEvalueStep, c as pairedEvalueSequence, i as PairedEvalueSequence, r as PairedEvalueOptions } from "../sequential-CYwq6Ff_.js";
|
|
7
|
+
//#region src/experiment/ast.d.ts
|
|
8
|
+
/** JSON-serializable value — everything a sealed node may carry. */
|
|
9
|
+
type JsonValue = string | number | boolean | null | JsonValue[] | {
|
|
10
|
+
[k: string]: JsonValue;
|
|
11
|
+
};
|
|
12
|
+
/** Evidence row shape. Fields are addressed by dot-separated paths. */
|
|
13
|
+
type EvidenceRecord = Record<string, unknown>;
|
|
14
|
+
/** A decision rule's branches did not cover the evidence. */
|
|
15
|
+
declare class DecisionTableNotTotalError extends ValidationError {}
|
|
16
|
+
/**
|
|
17
|
+
* Closed-key comparison over a declared record schema. The only leaf node.
|
|
18
|
+
* `field` is a dot-separated path into an evidence record.
|
|
19
|
+
*/
|
|
20
|
+
type Predicate = {
|
|
21
|
+
kind: 'compare';
|
|
22
|
+
field: string;
|
|
23
|
+
op: 'eq' | 'ne' | 'lt' | 'lte' | 'gt' | 'gte';
|
|
24
|
+
value: JsonValue;
|
|
25
|
+
} | {
|
|
26
|
+
kind: 'in';
|
|
27
|
+
field: string;
|
|
28
|
+
values: JsonValue[];
|
|
29
|
+
} | {
|
|
30
|
+
kind: 'all';
|
|
31
|
+
of: Predicate[];
|
|
32
|
+
} | {
|
|
33
|
+
kind: 'any';
|
|
34
|
+
of: Predicate[];
|
|
35
|
+
} | {
|
|
36
|
+
kind: 'not';
|
|
37
|
+
of: Predicate;
|
|
38
|
+
};
|
|
39
|
+
/** Read a dot-separated field path. Missing segments yield `undefined`. */
|
|
40
|
+
declare function readField(record: EvidenceRecord, path: string): unknown;
|
|
41
|
+
/** Evaluate a predicate against one evidence record. */
|
|
42
|
+
declare function evaluatePredicate(predicate: Predicate, record: EvidenceRecord): boolean;
|
|
43
|
+
/**
|
|
44
|
+
* One monotone funnel stage. `waives` names substrate admission conditions
|
|
45
|
+
* deliberately NOT applied, so a waiver is registered, never implicit.
|
|
46
|
+
*/
|
|
47
|
+
interface AdmissionStage {
|
|
48
|
+
id: string;
|
|
49
|
+
keep: Predicate;
|
|
50
|
+
waives?: string[];
|
|
51
|
+
}
|
|
52
|
+
/** A partition is a set reported separately and never pooled. */
|
|
53
|
+
interface AdmissionPartition {
|
|
54
|
+
id: string;
|
|
55
|
+
/** Stage whose dropped rows this partition draws from. */
|
|
56
|
+
from: string;
|
|
57
|
+
keep: Predicate;
|
|
58
|
+
pooling: 'never';
|
|
59
|
+
}
|
|
60
|
+
/** Declarative row-admission funnel. Stages only remove rows. */
|
|
61
|
+
interface AdmissionRule {
|
|
62
|
+
population: string;
|
|
63
|
+
stages: AdmissionStage[];
|
|
64
|
+
partitions?: AdmissionPartition[];
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* Deterministic subset selection. `reads` is the closed field set the rule
|
|
68
|
+
* may touch — outcome-contaminated selection is unrepresentable because an
|
|
69
|
+
* outcome field is simply not in the list.
|
|
70
|
+
*/
|
|
71
|
+
type SelectionRule = {
|
|
72
|
+
kind: 'round-robin';
|
|
73
|
+
groupBy: string;
|
|
74
|
+
groupOrder: 'lex-asc';
|
|
75
|
+
withinOrder: {
|
|
76
|
+
field: string;
|
|
77
|
+
dir: 'asc' | 'desc';
|
|
78
|
+
};
|
|
79
|
+
take: number;
|
|
80
|
+
reads: string[];
|
|
81
|
+
} | {
|
|
82
|
+
kind: 'filter-of';
|
|
83
|
+
/** Name of the sealed subset or selection this rule filters. */
|
|
84
|
+
base: string;
|
|
85
|
+
keep: Predicate;
|
|
86
|
+
order: {
|
|
87
|
+
field: string;
|
|
88
|
+
dir: 'asc' | 'desc';
|
|
89
|
+
};
|
|
90
|
+
};
|
|
91
|
+
/**
|
|
92
|
+
* Execute a selection rule.
|
|
93
|
+
*
|
|
94
|
+
* Round-robin walks groups in lexicographic order and takes ids in
|
|
95
|
+
* within-group order until `take` ids are chosen. Filter-of keeps the ids of
|
|
96
|
+
* `bases[rule.base]` whose record satisfies the predicate, in the registered
|
|
97
|
+
* order. Ids absent from `records` are evaluated on their id alone (fields
|
|
98
|
+
* derived from the id via `idFields`), so a sealed base outlives its source
|
|
99
|
+
* records.
|
|
100
|
+
*/
|
|
101
|
+
declare function runSelectionRule(rule: SelectionRule, records: readonly EvidenceRecord[], options: {
|
|
102
|
+
/** Field carrying a row's identity. */
|
|
103
|
+
idField: string;
|
|
104
|
+
/** Sealed or previously-computed subsets, by name. */
|
|
105
|
+
bases?: Record<string, readonly string[]>;
|
|
106
|
+
/** Derive predicate-readable fields from a bare id when its record is absent. */
|
|
107
|
+
idFields?: (id: string) => EvidenceRecord;
|
|
108
|
+
}): string[];
|
|
109
|
+
/** A named set of row identities, built from arm rows and an event predicate. */
|
|
110
|
+
type SetExpr = {
|
|
111
|
+
kind: 'rows-where';
|
|
112
|
+
arm: string;
|
|
113
|
+
event: Predicate;
|
|
114
|
+
} | {
|
|
115
|
+
kind: 'intersect';
|
|
116
|
+
of: SetExpr[];
|
|
117
|
+
};
|
|
118
|
+
/**
|
|
119
|
+
* What the experiment measures. Every estimand names the fields it reads, so
|
|
120
|
+
* the sealed tree records the full data dependency of the number.
|
|
121
|
+
*/
|
|
122
|
+
type Estimand = {
|
|
123
|
+
kind: 'rate';
|
|
124
|
+
event: Predicate;
|
|
125
|
+
over: 'rollouts';
|
|
126
|
+
} | {
|
|
127
|
+
kind: 'rate-at-least-once';
|
|
128
|
+
event: Predicate;
|
|
129
|
+
groupBy: string;
|
|
130
|
+
} | {
|
|
131
|
+
kind: 'paired-mean-diff';
|
|
132
|
+
armField: string;
|
|
133
|
+
treatment: string;
|
|
134
|
+
control: string;
|
|
135
|
+
pairBy: string;
|
|
136
|
+
value: string;
|
|
137
|
+
/** A pair one arm did not answer contributes a difference of exactly zero. */
|
|
138
|
+
missing: 'zero-diff';
|
|
139
|
+
} | {
|
|
140
|
+
kind: 'set-ratio';
|
|
141
|
+
armField: string;
|
|
142
|
+
idField: string;
|
|
143
|
+
numerator: SetExpr;
|
|
144
|
+
denominator: SetExpr;
|
|
145
|
+
};
|
|
146
|
+
interface EstimandResult {
|
|
147
|
+
value: number;
|
|
148
|
+
numerator: number;
|
|
149
|
+
denominator: number;
|
|
150
|
+
}
|
|
151
|
+
/** Compute an estimand over evidence rows. Pure; reads only registered fields. */
|
|
152
|
+
declare function computeEstimand(estimand: Estimand, rows: readonly EvidenceRecord[]): EstimandResult;
|
|
153
|
+
/** How uncertainty is computed. The seed is part of the registration. */
|
|
154
|
+
type IntervalSpec = {
|
|
155
|
+
kind: 'cluster-bootstrap';
|
|
156
|
+
clusterBy: string;
|
|
157
|
+
resamples: number;
|
|
158
|
+
seed: number;
|
|
159
|
+
level: number;
|
|
160
|
+
method: 'percentile';
|
|
161
|
+
} | {
|
|
162
|
+
kind: 'clopper-pearson';
|
|
163
|
+
level: number;
|
|
164
|
+
};
|
|
165
|
+
interface ComputedInterval {
|
|
166
|
+
lower: number;
|
|
167
|
+
upper: number;
|
|
168
|
+
level: number;
|
|
169
|
+
}
|
|
170
|
+
/**
|
|
171
|
+
* Execute an interval spec.
|
|
172
|
+
*
|
|
173
|
+
* Cluster-bootstrap resamples whole clusters of the per-row `value` field and
|
|
174
|
+
* takes percentile bounds of the pooled mean. Clopper-Pearson computes the
|
|
175
|
+
* exact binomial interval and requires `successes`/`trials` evidence instead
|
|
176
|
+
* of rows.
|
|
177
|
+
*/
|
|
178
|
+
declare function computeInterval(spec: IntervalSpec, evidence: {
|
|
179
|
+
kind: 'rows';
|
|
180
|
+
rows: readonly EvidenceRecord[];
|
|
181
|
+
value: string;
|
|
182
|
+
} | {
|
|
183
|
+
kind: 'binomial';
|
|
184
|
+
successes: number;
|
|
185
|
+
trials: number;
|
|
186
|
+
}): ComputedInterval;
|
|
187
|
+
/** Decision guards read only named derived quantities — never raw rows. */
|
|
188
|
+
type Condition = {
|
|
189
|
+
kind: 'interval-excludes-zero';
|
|
190
|
+
interval: string;
|
|
191
|
+
sign: 'positive' | 'negative';
|
|
192
|
+
} | {
|
|
193
|
+
kind: 'interval-includes-zero';
|
|
194
|
+
interval: string;
|
|
195
|
+
} | {
|
|
196
|
+
kind: 'quantity-threshold';
|
|
197
|
+
quantity: string;
|
|
198
|
+
op: 'gte' | 'lte' | 'gt' | 'lt';
|
|
199
|
+
value: number;
|
|
200
|
+
} | {
|
|
201
|
+
kind: 'obligation-met';
|
|
202
|
+
obligation: string;
|
|
203
|
+
} | {
|
|
204
|
+
kind: 'all';
|
|
205
|
+
of: Condition[];
|
|
206
|
+
} | {
|
|
207
|
+
kind: 'any';
|
|
208
|
+
of: Condition[];
|
|
209
|
+
} | {
|
|
210
|
+
kind: 'not';
|
|
211
|
+
of: Condition;
|
|
212
|
+
};
|
|
213
|
+
/** The named quantities a decision rule may read. Nothing else reaches it. */
|
|
214
|
+
interface DerivedQuantities {
|
|
215
|
+
intervals: Record<string, {
|
|
216
|
+
lower: number;
|
|
217
|
+
upper: number;
|
|
218
|
+
}>;
|
|
219
|
+
quantities: Record<string, number>;
|
|
220
|
+
obligationsMet: Record<string, boolean>;
|
|
221
|
+
}
|
|
222
|
+
declare function evaluateCondition(condition: Condition, evidence: DerivedQuantities): boolean;
|
|
223
|
+
interface DecisionBranch {
|
|
224
|
+
when: Condition;
|
|
225
|
+
verdict: string;
|
|
226
|
+
report: string[];
|
|
227
|
+
}
|
|
228
|
+
/**
|
|
229
|
+
* Ordered decision table: the first branch whose condition holds fires, and a
|
|
230
|
+
* table no branch matches throws — a non-total registration is a defect, not
|
|
231
|
+
* an implicit verdict. `report-only` registers the ABSENCE of a verdict
|
|
232
|
+
* branch: the estimate and the per-row table are the finding, and the
|
|
233
|
+
* registered meaning prose rides as non-executable interpretation data.
|
|
234
|
+
*/
|
|
235
|
+
type DecisionRule = {
|
|
236
|
+
kind: 'table';
|
|
237
|
+
branches: DecisionBranch[];
|
|
238
|
+
} | {
|
|
239
|
+
kind: 'report-only';
|
|
240
|
+
estimands: string[];
|
|
241
|
+
intervals: string[];
|
|
242
|
+
perRow: string[];
|
|
243
|
+
interpretation?: {
|
|
244
|
+
onQualitative: string;
|
|
245
|
+
consequence: string;
|
|
246
|
+
}[];
|
|
247
|
+
};
|
|
248
|
+
interface DecisionOutcome {
|
|
249
|
+
verdict: string;
|
|
250
|
+
report: string[];
|
|
251
|
+
}
|
|
252
|
+
declare function executeDecisionRule(rule: DecisionRule, evidence: DerivedQuantities): DecisionOutcome;
|
|
253
|
+
/** A registered control that must exist before a class of verdicts is read. */
|
|
254
|
+
interface Obligation {
|
|
255
|
+
id: string;
|
|
256
|
+
appliesToVerdicts: string[];
|
|
257
|
+
control: string;
|
|
258
|
+
}
|
|
259
|
+
/** Pre-spend design checks. Each returns pass/fail plus its evidence. */
|
|
260
|
+
type ValidityGate = {
|
|
261
|
+
kind: 'oracle-determinism';
|
|
262
|
+
unit: 'suite' | 'assertion';
|
|
263
|
+
replicates: number;
|
|
264
|
+
maxFlipRate: number;
|
|
265
|
+
} | {
|
|
266
|
+
kind: 'population-reproducibility';
|
|
267
|
+
joinOn: string;
|
|
268
|
+
compare: string[];
|
|
269
|
+
maxChangedRows: number;
|
|
270
|
+
} | {
|
|
271
|
+
kind: 'provenance-assertion';
|
|
272
|
+
claim: Predicate;
|
|
273
|
+
} | {
|
|
274
|
+
kind: 'power-floor';
|
|
275
|
+
target: number;
|
|
276
|
+
effectGrid: number[];
|
|
277
|
+
sim: {
|
|
278
|
+
trials: number;
|
|
279
|
+
resamples: number;
|
|
280
|
+
seed: number;
|
|
281
|
+
};
|
|
282
|
+
} | {
|
|
283
|
+
kind: 'identity';
|
|
284
|
+
field: 'served-model';
|
|
285
|
+
op: 'basename-eq';
|
|
286
|
+
onFail: 'abort';
|
|
287
|
+
};
|
|
288
|
+
interface GateResult {
|
|
289
|
+
id: string;
|
|
290
|
+
passed: boolean;
|
|
291
|
+
evidence: JsonValue;
|
|
292
|
+
}
|
|
293
|
+
/**
|
|
294
|
+
* Replicate-flip counting over graded states. A state whose replicates split
|
|
295
|
+
* between pass and fail is flipping; its flip rate is the minority share.
|
|
296
|
+
*/
|
|
297
|
+
declare function evaluateOracleDeterminismGate(id: string, gate: Extract<ValidityGate, {
|
|
298
|
+
kind: 'oracle-determinism';
|
|
299
|
+
}>, repsByState: Record<string, readonly boolean[]>): GateResult;
|
|
300
|
+
/**
|
|
301
|
+
* Join two population snapshots on `joinOn` and compare the registered fields.
|
|
302
|
+
* Only rows present in both snapshots are compared; a presence change is a
|
|
303
|
+
* different failure and needs its own gate.
|
|
304
|
+
*/
|
|
305
|
+
declare function evaluatePopulationReproducibilityGate(id: string, gate: Extract<ValidityGate, {
|
|
306
|
+
kind: 'population-reproducibility';
|
|
307
|
+
}>, populations: {
|
|
308
|
+
left: readonly EvidenceRecord[];
|
|
309
|
+
right: readonly EvidenceRecord[];
|
|
310
|
+
}): GateResult;
|
|
311
|
+
/** The registered claim about provenance must hold on the provenance record. */
|
|
312
|
+
declare function evaluateProvenanceGate(id: string, gate: Extract<ValidityGate, {
|
|
313
|
+
kind: 'provenance-assertion';
|
|
314
|
+
}>, provenance: EvidenceRecord): GateResult;
|
|
315
|
+
/** Final path segment equality between the pinned and the served identity. */
|
|
316
|
+
declare function evaluateIdentityGate(id: string, _gate: Extract<ValidityGate, {
|
|
317
|
+
kind: 'identity';
|
|
318
|
+
}>, identities: {
|
|
319
|
+
pinned: string;
|
|
320
|
+
served: string;
|
|
321
|
+
}): GateResult;
|
|
322
|
+
/**
|
|
323
|
+
* The design's power curve must reach the registered target at some grid
|
|
324
|
+
* effect. The curve must cover the registered effect grid exactly — a curve
|
|
325
|
+
* computed on a different grid is different evidence and is refused.
|
|
326
|
+
*/
|
|
327
|
+
declare function evaluatePowerFloorGate(id: string, gate: Extract<ValidityGate, {
|
|
328
|
+
kind: 'power-floor';
|
|
329
|
+
}>, curve: readonly {
|
|
330
|
+
effect: number;
|
|
331
|
+
power: number;
|
|
332
|
+
}[]): GateResult;
|
|
333
|
+
/** Checks as prerequisites: any named gate failing refuses the spend. */
|
|
334
|
+
interface HaltRule {
|
|
335
|
+
when: {
|
|
336
|
+
kind: 'any-gate-failed';
|
|
337
|
+
gates: string[];
|
|
338
|
+
};
|
|
339
|
+
action: 'refuse-spend';
|
|
340
|
+
report: 'settling-n';
|
|
341
|
+
}
|
|
342
|
+
interface HaltOutcome {
|
|
343
|
+
fired: boolean;
|
|
344
|
+
action: 'refuse-spend' | null;
|
|
345
|
+
failedGates: string[];
|
|
346
|
+
}
|
|
347
|
+
declare function evaluateHaltRule(halt: HaltRule, gates: readonly GateResult[]): HaltOutcome;
|
|
348
|
+
/**
|
|
349
|
+
* Spend schedules as data. `ledger` names every entry the ceiling counts, so
|
|
350
|
+
* what the gate reads is registered, not improvised at gate time.
|
|
351
|
+
*/
|
|
352
|
+
type BudgetRule = {
|
|
353
|
+
kind: 'uniform-pass';
|
|
354
|
+
ceilingUsd: number;
|
|
355
|
+
maxPasses: number;
|
|
356
|
+
projection: 'last-pass-cost';
|
|
357
|
+
costSource: 'priced-per-call';
|
|
358
|
+
ledger: {
|
|
359
|
+
id: string;
|
|
360
|
+
usd: number;
|
|
361
|
+
}[];
|
|
362
|
+
partialPass: 'report-never-lift';
|
|
363
|
+
} | {
|
|
364
|
+
kind: 'n-ladder';
|
|
365
|
+
steps: number[];
|
|
366
|
+
ceilingUsd: number;
|
|
367
|
+
projection: 'unit-cost-times-rows-times-n';
|
|
368
|
+
onExhaust: 'refuse-report-projection';
|
|
369
|
+
};
|
|
370
|
+
interface UniformPassDecision {
|
|
371
|
+
pass: number;
|
|
372
|
+
cumulativeBefore: number;
|
|
373
|
+
projected: number;
|
|
374
|
+
go: boolean;
|
|
375
|
+
}
|
|
376
|
+
interface UniformPassSchedule {
|
|
377
|
+
decisions: UniformPassDecision[];
|
|
378
|
+
uniformN: number;
|
|
379
|
+
}
|
|
380
|
+
/**
|
|
381
|
+
* Execute the uniform-pass schedule against measured pass costs. Pass 1 always
|
|
382
|
+
* runs; each later pass runs only when the cumulative spend plus the last
|
|
383
|
+
* measured pass cost stays at or under the registered ceiling. The registered
|
|
384
|
+
* ledger is the pre-spend the ceiling counts.
|
|
385
|
+
*/
|
|
386
|
+
declare function runUniformPassBudget(rule: Extract<BudgetRule, {
|
|
387
|
+
kind: 'uniform-pass';
|
|
388
|
+
}>, measuredPassCosts: readonly number[]): UniformPassSchedule;
|
|
389
|
+
interface NLadderProjection {
|
|
390
|
+
chosenN: number | null;
|
|
391
|
+
projections: {
|
|
392
|
+
n: number;
|
|
393
|
+
projectedUsd: number;
|
|
394
|
+
affordable: boolean;
|
|
395
|
+
}[];
|
|
396
|
+
refusal: null | {
|
|
397
|
+
onExhaust: 'refuse-report-projection';
|
|
398
|
+
reason: string;
|
|
399
|
+
};
|
|
400
|
+
}
|
|
401
|
+
/**
|
|
402
|
+
* Walk the registered n-ladder and pick the first affordable step. When no
|
|
403
|
+
* step fits the ceiling, the rule refuses and reports the projection instead
|
|
404
|
+
* of shrinking the row set — "never subset rows" is the registered invariant.
|
|
405
|
+
*/
|
|
406
|
+
declare function projectNLadderBudget(rule: Extract<BudgetRule, {
|
|
407
|
+
kind: 'n-ladder';
|
|
408
|
+
}>, measured: {
|
|
409
|
+
unitCostUsd: number;
|
|
410
|
+
rows: number;
|
|
411
|
+
}): NLadderProjection;
|
|
412
|
+
/** Arm budget matching as a refusal, not prose. Verified by `verifyMatchedBudgets`. */
|
|
413
|
+
interface MatchedBudgetRule {
|
|
414
|
+
measure: 'realized-tokens';
|
|
415
|
+
tolerance: number;
|
|
416
|
+
onFail: 'refuse-contrast';
|
|
417
|
+
}
|
|
418
|
+
/** Carrier faults are reissued; model outcomes stand. Closed enumeration. */
|
|
419
|
+
interface ReissuePolicy {
|
|
420
|
+
carrierEvents: ('http-status' | 'transport-error' | 'deadline' | 'empty-content')[];
|
|
421
|
+
modelOutcomesStand: true;
|
|
422
|
+
maxIssues: number;
|
|
423
|
+
}
|
|
424
|
+
type ReissueVerdict = 'reissue' | 'stands' | 'exhausted';
|
|
425
|
+
/**
|
|
426
|
+
* Classify one rollout event under the registered reissue policy. A carrier
|
|
427
|
+
* event within the issue budget is reissued; a model outcome always stands;
|
|
428
|
+
* a carrier event past `maxIssues` is exhausted and reported, never retried.
|
|
429
|
+
*/
|
|
430
|
+
declare function classifyReissue(policy: ReissuePolicy, event: string, issuesSoFar: number): ReissueVerdict;
|
|
431
|
+
//#endregion
|
|
432
|
+
//#region src/experiment/budget.d.ts
|
|
433
|
+
/** Arms whose realized budgets diverge past the registered tolerance. */
|
|
434
|
+
declare class MatchedBudgetError extends CaptureIntegrityError {}
|
|
435
|
+
interface ArmRealizedBudget {
|
|
436
|
+
armId: string;
|
|
437
|
+
/** Realized prompt + completion tokens the arm actually consumed. */
|
|
438
|
+
realizedTokens: number;
|
|
439
|
+
}
|
|
440
|
+
interface MatchedBudgetVerdict {
|
|
441
|
+
rule: MatchedBudgetRule;
|
|
442
|
+
arms: ArmRealizedBudget[];
|
|
443
|
+
/** (max - min) / max over all arms; 0 when every arm spent nothing. */
|
|
444
|
+
maxRelativeGap: number;
|
|
445
|
+
/** The two arms with the widest gap; null when fewer than two arms. */
|
|
446
|
+
widestPair: [string, string] | null;
|
|
447
|
+
matched: boolean;
|
|
448
|
+
/** Populated exactly when `matched` is false. */
|
|
449
|
+
refusal: null | {
|
|
450
|
+
onFail: 'refuse-contrast';
|
|
451
|
+
reason: string;
|
|
452
|
+
};
|
|
453
|
+
}
|
|
454
|
+
/**
|
|
455
|
+
* Compare realized per-arm spend under the registered tolerance. Requires at
|
|
456
|
+
* least two arms — a single arm has nothing to match against. Negative or
|
|
457
|
+
* non-finite token counts are refused as evidence corruption, not compared.
|
|
458
|
+
*/
|
|
459
|
+
declare function verifyMatchedBudgets(rule: MatchedBudgetRule, arms: readonly ArmRealizedBudget[]): MatchedBudgetVerdict;
|
|
460
|
+
/** Throw the refusal for callers that gate the contrast on it. */
|
|
461
|
+
declare function assertMatchedBudgets(rule: MatchedBudgetRule, arms: readonly ArmRealizedBudget[]): MatchedBudgetVerdict;
|
|
462
|
+
//#endregion
|
|
463
|
+
//#region src/experiment/funnel.d.ts
|
|
464
|
+
/** A funnel stage gained rows, double-counted them, or failed to reconcile. */
|
|
465
|
+
declare class FunnelIntegrityError extends CaptureIntegrityError {}
|
|
466
|
+
interface FunnelStageCount {
|
|
467
|
+
id: string;
|
|
468
|
+
entering: number;
|
|
469
|
+
excluded: number;
|
|
470
|
+
remaining: number;
|
|
471
|
+
/** Named exclusion reasons summing to `excluded`, when the stage has them. */
|
|
472
|
+
exclusions?: Record<string, number>;
|
|
473
|
+
/** Substrate checks this stage deliberately does not apply. Registered, never implicit. */
|
|
474
|
+
waives?: string[];
|
|
475
|
+
}
|
|
476
|
+
interface FunnelPartitionCount {
|
|
477
|
+
id: string;
|
|
478
|
+
/** Stage whose excluded rows the partition draws from. */
|
|
479
|
+
from: string;
|
|
480
|
+
count: number;
|
|
481
|
+
/** A partition is reported separately and never pooled into the survivors. */
|
|
482
|
+
pooling: 'never';
|
|
483
|
+
}
|
|
484
|
+
interface ExperimentFunnel {
|
|
485
|
+
population: string;
|
|
486
|
+
input: number;
|
|
487
|
+
stages: FunnelStageCount[];
|
|
488
|
+
surviving: number;
|
|
489
|
+
partitions: FunnelPartitionCount[];
|
|
490
|
+
}
|
|
491
|
+
interface FunnelStageInput {
|
|
492
|
+
id: string;
|
|
493
|
+
/** Rows this stage removed. */
|
|
494
|
+
excluded: number;
|
|
495
|
+
exclusions?: Record<string, number>;
|
|
496
|
+
waives?: string[];
|
|
497
|
+
}
|
|
498
|
+
/**
|
|
499
|
+
* Build a funnel from counts and refuse anything non-monotone.
|
|
500
|
+
*
|
|
501
|
+
* Refusals: a negative count, a stage that gains rows (excluded < 0 is the
|
|
502
|
+
* only way to gain — `remaining = entering - excluded` by construction, so a
|
|
503
|
+
* gain cannot be smuggled in through `remaining`), named exclusions that do
|
|
504
|
+
* not sum to the stage total, and a partition drawing from an unknown stage
|
|
505
|
+
* or exceeding what that stage excluded.
|
|
506
|
+
*/
|
|
507
|
+
declare function buildFunnel(input: {
|
|
508
|
+
population: string;
|
|
509
|
+
input: number;
|
|
510
|
+
stages: FunnelStageInput[];
|
|
511
|
+
partitions?: {
|
|
512
|
+
id: string;
|
|
513
|
+
from: string;
|
|
514
|
+
count: number;
|
|
515
|
+
}[];
|
|
516
|
+
}): ExperimentFunnel;
|
|
517
|
+
/** A chain that does not add up is a broken denominator, so this throws. */
|
|
518
|
+
declare function assertFunnelReconciles(funnel: ExperimentFunnel): void;
|
|
519
|
+
interface AdmissionExecution {
|
|
520
|
+
funnel: ExperimentFunnel;
|
|
521
|
+
survivors: EvidenceRecord[];
|
|
522
|
+
/** Partition rows by partition id, reported separately and never pooled. */
|
|
523
|
+
partitionRows: Record<string, EvidenceRecord[]>;
|
|
524
|
+
}
|
|
525
|
+
/**
|
|
526
|
+
* Run a sealed admission rule over evidence records. Each stage keeps the rows
|
|
527
|
+
* its predicate accepts; partitions draw from the rows their source stage
|
|
528
|
+
* dropped. The result embeds the funnel, so the denominator chain and the
|
|
529
|
+
* surviving rows can never disagree.
|
|
530
|
+
*/
|
|
531
|
+
declare function executeAdmissionRule(rule: AdmissionRule, records: readonly EvidenceRecord[]): AdmissionExecution;
|
|
532
|
+
/**
|
|
533
|
+
* Chain two funnels whose boundary agrees: the second funnel's input must be
|
|
534
|
+
* exactly the first funnel's survivors. Anything else is a gap or an
|
|
535
|
+
* injection, and both are refused.
|
|
536
|
+
*/
|
|
537
|
+
declare function composeFunnels(first: ExperimentFunnel, second: ExperimentFunnel): ExperimentFunnel;
|
|
538
|
+
/**
|
|
539
|
+
* Text render of the chain. One row per stage; the reconciliation line at the
|
|
540
|
+
* bottom restates `input = surviving + excluded` so a reader can check the
|
|
541
|
+
* arithmetic without a tool.
|
|
542
|
+
*/
|
|
543
|
+
declare function renderFunnelTable(funnel: ExperimentFunnel): string;
|
|
544
|
+
//#endregion
|
|
545
|
+
//#region src/experiment/define.d.ts
|
|
546
|
+
/** A sealed experiment whose digest no longer matches its spec. */
|
|
547
|
+
declare class SealIntegrityError extends ValidationError {}
|
|
548
|
+
interface ArmSpec {
|
|
549
|
+
id: string;
|
|
550
|
+
role: 'treatment' | 'control';
|
|
551
|
+
/** Digest of the AgentProfile the arm runs under. */
|
|
552
|
+
profileDigest?: string;
|
|
553
|
+
/** Digest of the pinned execution policy the arm runs under. */
|
|
554
|
+
policyDigest?: string;
|
|
555
|
+
/** Registered arm pins (model, seed, step budget, ...) as data. */
|
|
556
|
+
pins?: Record<string, JsonValue>;
|
|
557
|
+
}
|
|
558
|
+
type OutcomeSpec = {
|
|
559
|
+
kind: 'binary';
|
|
560
|
+
/** Where the pass verdict comes from, e.g. 'injected-suite'. */
|
|
561
|
+
source?: string;
|
|
562
|
+
digestVerified?: boolean;
|
|
563
|
+
/** What counts as a pass, e.g. 'exit-0' or 'reward-file-contains-1'. */
|
|
564
|
+
pass?: string;
|
|
565
|
+
/** Errored rollouts stay in the denominator; dropping them is unregistered. */
|
|
566
|
+
droppedRollouts?: 'forbidden';
|
|
567
|
+
} | {
|
|
568
|
+
kind: 'bounded-score';
|
|
569
|
+
min: number;
|
|
570
|
+
max: number;
|
|
571
|
+
orientation: 'higher-is-better' | 'lower-is-better';
|
|
572
|
+
};
|
|
573
|
+
/**
|
|
574
|
+
* Per-rollout seed derivation as a closed source list. `arm` is not in the
|
|
575
|
+
* union, so arm-dependent seeding is unrepresentable.
|
|
576
|
+
*/
|
|
577
|
+
interface SeedDerivation {
|
|
578
|
+
from: ('seed' | 'rowId' | 'rolloutIndex')[];
|
|
579
|
+
}
|
|
580
|
+
interface ExperimentSpec {
|
|
581
|
+
id: string;
|
|
582
|
+
/** Human prose for the audit trail — never executable. */
|
|
583
|
+
hypothesis?: string;
|
|
584
|
+
arms: ArmSpec[];
|
|
585
|
+
outcome: OutcomeSpec;
|
|
586
|
+
admission?: AdmissionRule;
|
|
587
|
+
/** Named deterministic subset rules. */
|
|
588
|
+
selections?: Record<string, SelectionRule>;
|
|
589
|
+
/**
|
|
590
|
+
* Selection OUTPUTS pinned into the seal. A filter-of base resolves here, so
|
|
591
|
+
* reusing an earlier draw is registered and a re-draw is a new digest.
|
|
592
|
+
*/
|
|
593
|
+
sealedSubsets?: Record<string, string[]>;
|
|
594
|
+
estimands?: Record<string, Estimand>;
|
|
595
|
+
intervals?: Record<string, IntervalSpec>;
|
|
596
|
+
decision: DecisionRule;
|
|
597
|
+
obligations?: Obligation[];
|
|
598
|
+
/** Named pre-spend validity gates; the halt rule references these names. */
|
|
599
|
+
gates?: Record<string, ValidityGate>;
|
|
600
|
+
halt?: HaltRule;
|
|
601
|
+
budget?: BudgetRule;
|
|
602
|
+
matchedBudget?: MatchedBudgetRule;
|
|
603
|
+
reissue?: ReissuePolicy;
|
|
604
|
+
seedDerivation?: SeedDerivation;
|
|
605
|
+
seed?: number;
|
|
606
|
+
}
|
|
607
|
+
/**
|
|
608
|
+
* Validate every cross-reference inside a spec and freeze it.
|
|
609
|
+
*
|
|
610
|
+
* A decision condition may only read a registered interval, a registered
|
|
611
|
+
* estimand, or a registered obligation; a halt rule may only reference
|
|
612
|
+
* registered gates; a filter-of base must resolve to a registered selection
|
|
613
|
+
* or sealed subset. Anything else is refused here, before sealing.
|
|
614
|
+
*/
|
|
615
|
+
declare function defineExperiment(spec: ExperimentSpec): ExperimentSpec;
|
|
616
|
+
interface SealAmendment {
|
|
617
|
+
/** ISO8601 timestamp of the re-seal. */
|
|
618
|
+
at: string;
|
|
619
|
+
reason: string;
|
|
620
|
+
/** Blindness attestations: what was verifiably unseen when the amendment was made. */
|
|
621
|
+
blind: string[];
|
|
622
|
+
/** Digest of the spec after this amendment. */
|
|
623
|
+
digest: string;
|
|
624
|
+
}
|
|
625
|
+
interface SealedExperiment {
|
|
626
|
+
spec: ExperimentSpec;
|
|
627
|
+
/** sha256-content over the canonicalized spec — the current registration. */
|
|
628
|
+
digest: string;
|
|
629
|
+
algo: 'sha256-content';
|
|
630
|
+
sealedAt: string;
|
|
631
|
+
/** Digest of the original registration, before any amendment. */
|
|
632
|
+
initialDigest: string;
|
|
633
|
+
amendments: SealAmendment[];
|
|
634
|
+
}
|
|
635
|
+
/** Validate, canonicalize, and hash a spec into its registration. */
|
|
636
|
+
declare function sealExperiment(spec: ExperimentSpec, options?: {
|
|
637
|
+
sealedAt?: string;
|
|
638
|
+
}): Promise<SealedExperiment>;
|
|
639
|
+
/**
|
|
640
|
+
* Amend a sealed experiment. The current seal is verified first, the new spec
|
|
641
|
+
* is validated and re-hashed, and the amendment appends to the digest chain.
|
|
642
|
+
* There is no way to change what is decided without producing a new digest.
|
|
643
|
+
*/
|
|
644
|
+
declare function amendExperiment(sealed: SealedExperiment, amendment: {
|
|
645
|
+
spec: ExperimentSpec;
|
|
646
|
+
reason: string;
|
|
647
|
+
blind: string[];
|
|
648
|
+
at?: string;
|
|
649
|
+
}): Promise<SealedExperiment>;
|
|
650
|
+
/** True when the sealed digest still matches the spec it carries. */
|
|
651
|
+
declare function verifySealedExperiment(sealed: SealedExperiment): Promise<boolean>;
|
|
652
|
+
/** Evidence a gate executor consumes, discriminated to match the gate kind. */
|
|
653
|
+
type GateEvidence = {
|
|
654
|
+
kind: 'oracle-determinism';
|
|
655
|
+
repsByState: Record<string, readonly boolean[]>;
|
|
656
|
+
} | {
|
|
657
|
+
kind: 'population-reproducibility';
|
|
658
|
+
left: readonly EvidenceRecord[];
|
|
659
|
+
right: readonly EvidenceRecord[];
|
|
660
|
+
} | {
|
|
661
|
+
kind: 'provenance-assertion';
|
|
662
|
+
provenance: EvidenceRecord;
|
|
663
|
+
} | {
|
|
664
|
+
kind: 'identity';
|
|
665
|
+
pinned: string;
|
|
666
|
+
served: string;
|
|
667
|
+
} | {
|
|
668
|
+
kind: 'power-floor';
|
|
669
|
+
curve: readonly {
|
|
670
|
+
effect: number;
|
|
671
|
+
power: number;
|
|
672
|
+
}[];
|
|
673
|
+
};
|
|
674
|
+
/**
|
|
675
|
+
* Executors bound to one verified seal. Every method reads its rule from the
|
|
676
|
+
* sealed spec; evidence is the only argument anywhere.
|
|
677
|
+
*/
|
|
678
|
+
interface RegisteredExperiment {
|
|
679
|
+
readonly sealed: SealedExperiment;
|
|
680
|
+
/** Execute the registered decision rule on derived quantities. */
|
|
681
|
+
decide(evidence: DerivedQuantities): DecisionOutcome;
|
|
682
|
+
/** Run the registered admission funnel over evidence rows. */
|
|
683
|
+
admit(records: readonly EvidenceRecord[]): AdmissionExecution;
|
|
684
|
+
/** Run a registered selection rule. Filter-of bases resolve to sealed subsets. */
|
|
685
|
+
select(name: string, records: readonly EvidenceRecord[], options: {
|
|
686
|
+
idField: string;
|
|
687
|
+
}): string[];
|
|
688
|
+
/** Evaluate a registered validity gate on evidence of the matching kind. */
|
|
689
|
+
gate(name: string, evidence: GateEvidence): GateResult;
|
|
690
|
+
/** Evaluate the registered halt rule over gate results. */
|
|
691
|
+
halt(gates: readonly GateResult[]): HaltOutcome;
|
|
692
|
+
/** Execute the registered uniform-pass budget schedule. */
|
|
693
|
+
runUniformPassBudget(measuredPassCosts: readonly number[]): UniformPassSchedule;
|
|
694
|
+
/** Project the registered n-ladder budget. */
|
|
695
|
+
projectNLadderBudget(measured: {
|
|
696
|
+
unitCostUsd: number;
|
|
697
|
+
rows: number;
|
|
698
|
+
}): NLadderProjection;
|
|
699
|
+
/** Verify realized arm budgets under the registered matched-budget rule. */
|
|
700
|
+
matchedBudgets(arms: readonly ArmRealizedBudget[]): MatchedBudgetVerdict;
|
|
701
|
+
/** Compute a registered estimand over evidence rows. */
|
|
702
|
+
estimate(name: string, rows: readonly EvidenceRecord[]): EstimandResult;
|
|
703
|
+
/** Compute a registered interval spec. */
|
|
704
|
+
interval(name: string, evidence: {
|
|
705
|
+
kind: 'rows';
|
|
706
|
+
rows: readonly EvidenceRecord[];
|
|
707
|
+
value: string;
|
|
708
|
+
} | {
|
|
709
|
+
kind: 'binomial';
|
|
710
|
+
successes: number;
|
|
711
|
+
trials: number;
|
|
712
|
+
}): ComputedInterval;
|
|
713
|
+
}
|
|
714
|
+
/**
|
|
715
|
+
* Verify the seal and return executors bound to it. This is the module's only
|
|
716
|
+
* execution surface: a rule that is not in the sealed spec cannot run, and a
|
|
717
|
+
* rule that is cannot run differently.
|
|
718
|
+
*/
|
|
719
|
+
declare function openSealedExperiment(sealed: SealedExperiment): Promise<RegisteredExperiment>;
|
|
720
|
+
//#endregion
|
|
721
|
+
//#region src/experiment/power.d.ts
|
|
722
|
+
/** A design refused at configuration time, before any spend. */
|
|
723
|
+
declare class DesignRefusalError extends ValidationError {}
|
|
724
|
+
interface ClusteredPowerOptions {
|
|
725
|
+
/** Rows per independent cluster, e.g. [6, 3, 3, 2]. */
|
|
726
|
+
clusterSizes: number[];
|
|
727
|
+
/** Per-row effect grid: each effect is P(win) - P(loss) added to the base rates. */
|
|
728
|
+
effects: number[];
|
|
729
|
+
/** Deterministic seed for outcome draws and bootstrap resampling. */
|
|
730
|
+
seed: number;
|
|
731
|
+
/** Simulated experiments per effect. Default 2000. */
|
|
732
|
+
trials?: number;
|
|
733
|
+
/** Whole-cluster bootstrap draws per trial. Default 4000. */
|
|
734
|
+
resamples?: number;
|
|
735
|
+
/** Percentile interval level. Default 0.95. */
|
|
736
|
+
confidence?: number;
|
|
737
|
+
/** Sign-flip alpha the closed-form floor is checked against. Default 0.05. */
|
|
738
|
+
alpha?: number;
|
|
739
|
+
/** Power the design must reach at some grid effect. Default 0.8. */
|
|
740
|
+
targetPower?: number;
|
|
741
|
+
/** Base P(row favors treatment) with no effect. Default 0.10 (tie-heavy rows). */
|
|
742
|
+
baseWinRate?: number;
|
|
743
|
+
/** Base P(row favors control). Default 0.10. */
|
|
744
|
+
baseLossRate?: number;
|
|
745
|
+
/**
|
|
746
|
+
* Clusters whose rows carry outcome noise instead of signal: each row wins
|
|
747
|
+
* with `flipRate` and loses with `flipRate`, independent of the effect.
|
|
748
|
+
* Index into `clusterSizes`.
|
|
749
|
+
*/
|
|
750
|
+
noisyClusters?: {
|
|
751
|
+
index: number;
|
|
752
|
+
flipRate: number;
|
|
753
|
+
}[];
|
|
754
|
+
}
|
|
755
|
+
interface ClusteredPowerPoint {
|
|
756
|
+
effect: number;
|
|
757
|
+
power: number;
|
|
758
|
+
medianCiWidth: number;
|
|
759
|
+
}
|
|
760
|
+
interface SignFlipFloor {
|
|
761
|
+
/** Smallest achievable two-sided p: 2^(1-C) for C clusters. */
|
|
762
|
+
twoSidedP: number;
|
|
763
|
+
/** Smallest achievable one-sided p: 2^-C. */
|
|
764
|
+
oneSidedP: number;
|
|
765
|
+
alpha: number;
|
|
766
|
+
/** False when the cluster count can never certify at `alpha`, at any effect. */
|
|
767
|
+
certifiableAtAlpha: boolean;
|
|
768
|
+
/** Smallest cluster count whose two-sided floor is at or under `alpha`. */
|
|
769
|
+
minClustersForAlpha: number;
|
|
770
|
+
}
|
|
771
|
+
interface ClusteredPowerRefusal {
|
|
772
|
+
verdict: 'underpowered';
|
|
773
|
+
reasons: string[];
|
|
774
|
+
recommendation: string;
|
|
775
|
+
}
|
|
776
|
+
interface ClusteredPowerResult {
|
|
777
|
+
clusterCount: number;
|
|
778
|
+
totalRows: number;
|
|
779
|
+
trials: number;
|
|
780
|
+
resamples: number;
|
|
781
|
+
seed: number;
|
|
782
|
+
confidence: number;
|
|
783
|
+
targetPower: number;
|
|
784
|
+
curve: ClusteredPowerPoint[];
|
|
785
|
+
/** Maximum simulated power across the effect grid. */
|
|
786
|
+
maxPower: number;
|
|
787
|
+
signFlipFloor: SignFlipFloor;
|
|
788
|
+
/** True only when the sign-flip floor certifies AND simulation reaches target. */
|
|
789
|
+
adequate: boolean;
|
|
790
|
+
/** Populated exactly when `adequate` is false. The refusal lives in the artifact. */
|
|
791
|
+
refusal: ClusteredPowerRefusal | null;
|
|
792
|
+
}
|
|
793
|
+
/**
|
|
794
|
+
* Simulate the power of a whole-cluster percentile-bootstrap design and refuse
|
|
795
|
+
* a structure that cannot reach the target at any registered effect.
|
|
796
|
+
*/
|
|
797
|
+
declare function clusteredPower(options: ClusteredPowerOptions): ClusteredPowerResult;
|
|
798
|
+
/** Throw the refusal for callers that want configuration-time failure. */
|
|
799
|
+
declare function assertDesignAdequate(result: ClusteredPowerResult): void;
|
|
800
|
+
//#endregion
|
|
801
|
+
export { type AdmissionExecution, type AdmissionPartition, type AdmissionRule, type AdmissionStage, type ArmRealizedBudget, type ArmSpec, BOOTSTRAP_GATE_MIN_N, type BudgetRule, type ClusterBootstrapInterval, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, type ClusteredPowerOptions, type ClusteredPowerPoint, type ClusteredPowerRefusal, type ClusteredPowerResult, type ComputedInterval, type Condition, type DecisionBranch, type DecisionOutcome, type DecisionRule, DecisionTableNotTotalError, type DerivedQuantities, DesignRefusalError, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, type Estimand, type EstimandResult, type EvidenceRecord, type EvidenceVector, type Experiment, type ExperimentFunnel, type ExperimentRep, type ExperimentSpec, type ExperimentStats, ExperimentTracker, type ExperimentVerdict, FunnelIntegrityError, type FunnelPartitionCount, type FunnelStageCount, type FunnelStageInput, type GateEvidence, type GateResult, type HaltOutcome, type HaltRule, type HeldoutSignificance, type HeldoutSignificanceOptions, type HypothesisManifest, type HypothesisResult, type IntervalSpec, type JsonValue, MatchedBudgetError, type MatchedBudgetRule, type MatchedBudgetVerdict, type MatchedPair, type NLadderProjection, type Obligation, type OutcomeSpec, type PairArmsOptions, type PairArmsResult, type PairedArmRow, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedHoldout, type PowerPreflight, type PowerPreflightOptions, type Predicate, type PromotionPolicy, type ProportionInterval, type RegisteredExperiment, type ReissuePolicy, type ReissueVerdict, type SealAmendment, SealIntegrityError, type SealedExperiment, type SeedDerivation, type SelectionRule, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SetExpr, type SignFlipFloor, type SignedManifest, type UniformPassDecision, type UniformPassSchedule, type ValidityGate, amendExperiment, assertDesignAdequate, assertFunnelReconciles, assertMatchedBudgets, benjaminiHochberg, bonferroni, buildEvidenceVector, buildFunnel, canonicalize, classifyReissue, clusteredPairedBinary, clusteredPower, comparePairedArms, composeFunnels, computeEstimand, computeInterval, defineExperiment, eProcess, evaluateCondition, evaluateHaltRule, evaluateHypothesis, evaluateIdentityGate, evaluateOracleDeterminismGate, evaluatePopulationReproducibilityGate, evaluatePowerFloorGate, evaluatePredicate, evaluateProvenanceGate, executeAdmissionRule, executeDecisionRule, fileExperimentStore, hashJson, heldoutSignificance, holm, inMemoryExperimentStore, mcnemar, mcnemarPower, mcnemarRequiredN, mulberry32, openSealedExperiment, pairArms, pairHoldout, pairRunRecords, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, paretoPolicy, paretoSignificanceGate, powerPreflight, projectNLadderBudget, readField, renderFunnelTable, requiredPairedSampleSize, requiredSampleSize, runSelectionRule, runUniformPassBudget, sealExperiment, sequentialDecide, sequentialPairedGate, signManifest, verifyManifest, verifyMatchedBudgets, verifySealedExperiment, wilson };
|
|
802
|
+
//# sourceMappingURL=index.d.ts.map
|