@tangle-network/agent-eval 0.144.6 → 0.144.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +470 -88
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BKENp2s5.js} +662 -605
- package/dist/benchmark-command-BKENp2s5.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BEOkuvIg.js → benchmarks-BlPmjd88.js} +4 -4
- package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-BlPmjd88.js.map} +1 -1
- package/dist/campaign/index.d.ts +7 -5
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-CXsdyym7.js → campaign--HVSuvV0.js} +17 -301
- package/dist/campaign--HVSuvV0.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
- package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
- package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -8
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
- package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
- package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +214 -404
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +233 -649
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
- package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
- package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
- package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +2 -2
- package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
- package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +17 -7
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
- package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
- package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
- package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CkaI2ly4.js} +19 -654
- package/dist/skillopt-optimization-method-CkaI2ly4.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
- package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
- package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +2 -2
- package/dist/wire/index.js +1 -1
- package/docs/campaign-proposers.md +5 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +19 -2
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
- package/dist/benchmark-command-CQd78YHt.js.map +0 -1
- package/dist/campaign-CXsdyym7.js.map +0 -1
- package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
- package/dist/index-4XwggC10.d.ts.map +0 -1
- package/dist/index-BIL5vxxt.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-Dv5BiKLE.js.map +0 -1
- package/dist/replay-Krvb114g.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
- package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
- package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
- package/dist/types-XMVEdrE_.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
import { R as Scenario, h as GateContext, p as Gate, v as GateResult } from "./types-DYuNHo9R.js";
|
|
2
|
+
import { y as PairedBootstrapResult } from "./statistics-D6Uebe_4.js";
|
|
3
|
+
import { i as PairedMcNemarEvidence, r as PairedDecisionStatistic, t as PairedDecisionMethod } from "./paired-promotion-decision-B6zJ3gYM.js";
|
|
4
|
+
//#region src/campaign/gates/power-preflight.d.ts
|
|
5
|
+
/**
|
|
6
|
+
* Power preflight — "can this budget detect the effect you are hunting?"
|
|
7
|
+
*
|
|
8
|
+
* The failure it prevents (measured, twice): a live prompt-improvement campaign ran
|
|
9
|
+
* 333 sandbox cells over 5.6 hours and produced a +0.08 holdout lift the ship gate
|
|
10
|
+
* (paired bootstrap, CI.low > 0.05) could not distinguish from zero — because at
|
|
11
|
+
* that holdout size and worker variance the MINIMUM DETECTABLE lift was larger than
|
|
12
|
+
* any effect a prompt change plausibly produces. The budget was spent learning what
|
|
13
|
+
* a 30-second calculation on the baseline cells already knew. No eval framework we
|
|
14
|
+
* know of surfaces this; every underpowered improvement run everywhere ends in an
|
|
15
|
+
* uninformative "hold".
|
|
16
|
+
*
|
|
17
|
+
* Model: the ship rule is `CI.low(paired Δ) > deltaThreshold`. Approximating the
|
|
18
|
+
* bootstrap CI as normal, `CI.low ≈ effect − z·sd_Δ/√n`, so the smallest shippable
|
|
19
|
+
* true effect is `MDE = deltaThreshold + z·sd_Δ/√n`. The paired-delta SD is unknown
|
|
20
|
+
* before the candidate exists; we bound it by the zero-correlation case
|
|
21
|
+
* `sd_Δ ≤ √2·sd_baseline` — a CONSERVATIVE (upper) MDE, which is the correct
|
|
22
|
+
* direction for a warning. Pairing is per cell (`scenario:rep`), so reps multiply n.
|
|
23
|
+
*
|
|
24
|
+
* Standalone by design: feed it any baseline composites (a `gate:'none'` run, a
|
|
25
|
+
* live-proof table) BEFORE budgeting the real search; `selfImprove` also attaches
|
|
26
|
+
* it to every result and warns when the run was structurally unable to ship.
|
|
27
|
+
*/
|
|
28
|
+
interface PowerPreflightOptions {
|
|
29
|
+
/** Per-cell baseline composites on the HOLDOUT scenarios (one per scenario:rep cell). */
|
|
30
|
+
baselineComposites: number[];
|
|
31
|
+
/** Paired observations the budgeted comparison will produce
|
|
32
|
+
* (holdout scenarios × reps). Defaults to `baselineComposites.length`. */
|
|
33
|
+
pairedN?: number;
|
|
34
|
+
/** The ship gate's effect-size threshold. Default 0.05 (defaultProductionGate). */
|
|
35
|
+
deltaThreshold?: number;
|
|
36
|
+
/** CI confidence the gate uses. Default 0.95. */
|
|
37
|
+
confidence?: number;
|
|
38
|
+
/** True when the holdout is scored by the SAME judge/scorer family as the gate
|
|
39
|
+
* (selfImprove's default composition — one judge scores everything). Under a
|
|
40
|
+
* shared channel, raising paired n reduces only the IDIOSYNCRATIC noise share;
|
|
41
|
+
* systematic judge bias is untouched, so the MDE here is a lower bound and the
|
|
42
|
+
* only full debiaser is an independent second scoring channel
|
|
43
|
+
* (recursive-self-improvement S1c, closed form in EXP-023 P0). Default false. */
|
|
44
|
+
sharedScorerChannel?: boolean;
|
|
45
|
+
}
|
|
46
|
+
interface PowerPreflight {
|
|
47
|
+
/** Paired observations the comparison will have. */
|
|
48
|
+
n: number;
|
|
49
|
+
/** Baseline per-cell composite standard deviation (the variance the effect must beat). */
|
|
50
|
+
sd: number;
|
|
51
|
+
/** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */
|
|
52
|
+
mde: number;
|
|
53
|
+
/** Baseline holdout composite mean. */
|
|
54
|
+
baselineMean: number;
|
|
55
|
+
/** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */
|
|
56
|
+
headroom: number;
|
|
57
|
+
/** True when even the largest achievable effect (headroom) is below the MDE —
|
|
58
|
+
* the run is structurally unable to ship regardless of proposal quality.
|
|
59
|
+
* Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */
|
|
60
|
+
underpowered: boolean;
|
|
61
|
+
/** True when composites look [0,1]-scaled; headroom/underpowered are only
|
|
62
|
+
* meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */
|
|
63
|
+
scaleAssumed: boolean;
|
|
64
|
+
deltaThreshold: number;
|
|
65
|
+
confidence: number;
|
|
66
|
+
/** Set when the holdout shares the gate's scoring channel: more cells cannot
|
|
67
|
+
* buy back systematic judge bias — treat the MDE as a lower bound. */
|
|
68
|
+
sharedChannelCaveat?: string;
|
|
69
|
+
/** One actionable sentence for humans and logs. */
|
|
70
|
+
recommendation: string;
|
|
71
|
+
}
|
|
72
|
+
/** Estimate the minimum detectable lift a paired-holdout improvement run can
|
|
73
|
+
* ship at a given budget, from the baseline holdout composites — call it BEFORE
|
|
74
|
+
* spending a search to learn whether the effect you are hunting is even
|
|
75
|
+
* observable at this holdout size and worker variance. */
|
|
76
|
+
declare function powerPreflight(opts: PowerPreflightOptions): PowerPreflight;
|
|
77
|
+
//#endregion
|
|
78
|
+
//#region src/pareto.d.ts
|
|
79
|
+
/**
|
|
80
|
+
* Pareto frontier — multi-objective optimization over candidate runs.
|
|
81
|
+
*
|
|
82
|
+
* Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
|
|
83
|
+
* trading off (cost, latency, quality) or (passRate, tokenBudget,
|
|
84
|
+
* ttfb), you rarely have a single "winner" — you have a set of
|
|
85
|
+
* non-dominated candidates. This module exposes:
|
|
86
|
+
*
|
|
87
|
+
* - `paretoFrontier`: filter a set of candidates to the non-dominated ones
|
|
88
|
+
* - `dominates`: does A dominate B across all objectives?
|
|
89
|
+
*
|
|
90
|
+
* Each objective is declared with a direction: 'maximize' (higher=better)
|
|
91
|
+
* or 'minimize' (lower=better). Candidates are any object; pass an
|
|
92
|
+
* `objective(candidate)` accessor.
|
|
93
|
+
*/
|
|
94
|
+
type Direction = 'maximize' | 'minimize';
|
|
95
|
+
interface Objective<T> {
|
|
96
|
+
/** Stable label used in reports. */
|
|
97
|
+
name: string;
|
|
98
|
+
direction: Direction;
|
|
99
|
+
value: (candidate: T) => number;
|
|
100
|
+
}
|
|
101
|
+
interface ParetoResult<T> {
|
|
102
|
+
frontier: T[];
|
|
103
|
+
dominated: T[];
|
|
104
|
+
/** Index map: frontier[i] dominates each of dominatedBy[i]. */
|
|
105
|
+
dominanceMap: Array<{
|
|
106
|
+
dominator: T;
|
|
107
|
+
dominated: T[];
|
|
108
|
+
}>;
|
|
109
|
+
}
|
|
110
|
+
/** Does candidate A weakly dominate B — strictly better on at least one objective and no worse on any? */
|
|
111
|
+
declare function dominates<T>(a: T, b: T, objectives: Objective<T>[]): boolean;
|
|
112
|
+
/**
|
|
113
|
+
* Compute the non-dominated frontier. Candidates with NaN/Infinity on any
|
|
114
|
+
* objective are excluded (can't rank them). A candidate enters the frontier
|
|
115
|
+
* iff no other candidate dominates it.
|
|
116
|
+
*/
|
|
117
|
+
declare function paretoFrontier<T>(candidates: T[], objectives: Objective<T>[]): ParetoResult<T>;
|
|
118
|
+
/**
|
|
119
|
+
* Weighted-sum scalarisation. Use as a tie-break / single-winner selector
|
|
120
|
+
* when callers don't want to consume a frontier. Each objective contributes
|
|
121
|
+
* its normalised value (0..1 via min-max across the candidate pool) times
|
|
122
|
+
* its weight; missing weights default to 1/N.
|
|
123
|
+
*
|
|
124
|
+
* Direction is honoured automatically — `minimize` axes have their values
|
|
125
|
+
* inverted before scaling so "higher scalar = better" always holds.
|
|
126
|
+
*/
|
|
127
|
+
declare function scalarScore<T>(candidates: T[], objectives: Objective<T>[], options?: {
|
|
128
|
+
weights?: Partial<Record<string, number>>;
|
|
129
|
+
}): Array<{
|
|
130
|
+
candidate: T;
|
|
131
|
+
score: number;
|
|
132
|
+
}>;
|
|
133
|
+
/**
|
|
134
|
+
* NSGA-II crowding distance — secondary sort for ties on the frontier.
|
|
135
|
+
*
|
|
136
|
+
* When the Pareto front collapses to a single point (or many candidates tie
|
|
137
|
+
* on dominance), naive selection picks arbitrarily and the population
|
|
138
|
+
* degenerates over generations. NSGA-II preserves diversity by preferring
|
|
139
|
+
* candidates with more empty space around them on the frontier.
|
|
140
|
+
*
|
|
141
|
+
* Returns an array of `{ candidate, distance }` in the SAME order as the
|
|
142
|
+
* input. Higher distance = more isolated = should be preferred when
|
|
143
|
+
* preserving diversity.
|
|
144
|
+
*/
|
|
145
|
+
declare function crowdingDistance<T>(candidates: T[], objectives: Objective<T>[]): Array<{
|
|
146
|
+
candidate: T;
|
|
147
|
+
distance: number;
|
|
148
|
+
}>;
|
|
149
|
+
/**
|
|
150
|
+
* Pareto frontier with tie-break by crowding distance — the canonical
|
|
151
|
+
* NSGA-II selection step. Returns the frontier sorted by descending crowding
|
|
152
|
+
* distance so callers can `.slice(0, k)` to pick K diverse winners.
|
|
153
|
+
*/
|
|
154
|
+
declare function paretoFrontierWithCrowding<T>(candidates: T[], objectives: Objective<T>[]): Array<{
|
|
155
|
+
candidate: T;
|
|
156
|
+
distance: number;
|
|
157
|
+
}>;
|
|
158
|
+
//#endregion
|
|
159
|
+
//#region src/campaign/gates/promotion-policy.d.ts
|
|
160
|
+
/** Where an objective's per-cell scalar comes from. `composite` reads the
|
|
161
|
+
* judge's composite; `dimension` reads a named per-dimension score. */
|
|
162
|
+
type ObjectiveSource = {
|
|
163
|
+
kind: 'composite';
|
|
164
|
+
} | {
|
|
165
|
+
kind: 'dimension';
|
|
166
|
+
dimension: string;
|
|
167
|
+
};
|
|
168
|
+
interface PromotionObjective {
|
|
169
|
+
/** Stable label used in reports + `contributingGates`. */
|
|
170
|
+
name: string;
|
|
171
|
+
source: ObjectiveSource;
|
|
172
|
+
/** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients
|
|
173
|
+
* the paired delta so a positive bootstrap always means "candidate better". */
|
|
174
|
+
direction: Direction;
|
|
175
|
+
/** The good-direction paired-delta CI lower bound must EXCEED this to count
|
|
176
|
+
* as a significant gain on this axis. Interpreted in the judge's native
|
|
177
|
+
* scale. Default 0 (⇒ "confidently better"). */
|
|
178
|
+
gainThreshold?: number;
|
|
179
|
+
/** A floor breach (regression) is declared when the good-direction CI lower
|
|
180
|
+
* bound is below −floorTolerance, or when the exact small-sample test proves
|
|
181
|
+
* a drop past it. When omitted it auto-scales off observed magnitudes
|
|
182
|
+
* (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */
|
|
183
|
+
floorTolerance?: number;
|
|
184
|
+
}
|
|
185
|
+
/** Per-axis verdict from the good-direction paired bootstrap. */
|
|
186
|
+
type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs';
|
|
187
|
+
interface AxisEvidence {
|
|
188
|
+
name: string;
|
|
189
|
+
source: ObjectiveSource;
|
|
190
|
+
direction: Direction;
|
|
191
|
+
/** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):
|
|
192
|
+
* a positive value means the candidate is better on this axis.
|
|
193
|
+
*
|
|
194
|
+
* DIAGNOSTIC on a pass/fail axis: there the verdict is decided on Tango's
|
|
195
|
+
* score interval instead, because a percentile bootstrap over a three-atom
|
|
196
|
+
* delta lattice is not a valid interval at the nonzero margin `floorTolerance`
|
|
197
|
+
* and `gainThreshold` create. `ci` carries the interval that decided. */
|
|
198
|
+
bootstrap: PairedBootstrapResult;
|
|
199
|
+
/** Which paired statistic `bootstrap.low`/`.high` bracket. `'mean'` unless the
|
|
200
|
+
* caller asked for the median — on a pass/fail axis the median and its whole
|
|
201
|
+
* CI are pinned at 0 by tie domination and can see neither a gain nor a
|
|
202
|
+
* regression. `bootstrap.median` still carries the median point estimate. */
|
|
203
|
+
bootstrapStatistic: 'median' | 'mean';
|
|
204
|
+
/** The interval the axis verdict was actually decided on, good-direction and
|
|
205
|
+
* in the axis's native units. */
|
|
206
|
+
ci: {
|
|
207
|
+
low: number;
|
|
208
|
+
high: number;
|
|
209
|
+
};
|
|
210
|
+
/** Which estimator produced `ci`. */
|
|
211
|
+
decisionStatistic: PairedDecisionStatistic;
|
|
212
|
+
/** McNemar's exact evidence on a pass/fail axis; null otherwise. */
|
|
213
|
+
mcnemar: PairedMcNemarEvidence | null;
|
|
214
|
+
/** `ci` has zero width — no evidence in either direction, so the axis is
|
|
215
|
+
* neither improved nor regressed however the point estimate sits. */
|
|
216
|
+
indeterminate: boolean;
|
|
217
|
+
/** Paired observations contributing to this axis. */
|
|
218
|
+
n: number;
|
|
219
|
+
minimumRequired: number;
|
|
220
|
+
decisionMethod: PairedDecisionMethod;
|
|
221
|
+
gainThreshold: number;
|
|
222
|
+
floorTolerance: number;
|
|
223
|
+
verdict: AxisVerdict;
|
|
224
|
+
}
|
|
225
|
+
interface EvidenceVector {
|
|
226
|
+
/** One entry per objective — NOTHING averaged across axes. */
|
|
227
|
+
axes: AxisEvidence[];
|
|
228
|
+
/** Smallest paired n across axes that produced observations — the binding
|
|
229
|
+
* evidence-sufficiency constraint. 0 when no axis produced observations. */
|
|
230
|
+
minN: number;
|
|
231
|
+
/** Aggregate per-side cost from the gate context (a constraint input, not a
|
|
232
|
+
* CI axis — see the module header). */
|
|
233
|
+
cost: {
|
|
234
|
+
candidate: number;
|
|
235
|
+
baseline: number;
|
|
236
|
+
};
|
|
237
|
+
}
|
|
238
|
+
/** A promotion strategy: a pure function from the evidence vector to a verdict.
|
|
239
|
+
* Many policies can run over the same `EvidenceVector` and disagree — that's
|
|
240
|
+
* the point (competing strategies, shared evidence). */
|
|
241
|
+
type PromotionPolicy = (ev: EvidenceVector) => GateResult;
|
|
242
|
+
interface BuildEvidenceVectorOptions {
|
|
243
|
+
/** Minimum paired observations before an axis can claim significance; below
|
|
244
|
+
* it the axis is `few_runs`. The exact small-sample test may require more
|
|
245
|
+
* observations at the selected confidence. */
|
|
246
|
+
minProductiveRuns?: number;
|
|
247
|
+
/** Confidence level for every axis bootstrap. Default 0.95. */
|
|
248
|
+
confidence?: number;
|
|
249
|
+
/** Bootstrap resamples. Default 2000. */
|
|
250
|
+
resamples?: number;
|
|
251
|
+
/** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */
|
|
252
|
+
seed?: number;
|
|
253
|
+
/** Paired statistic every axis CI is computed on. Default `'mean'` — see
|
|
254
|
+
* {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */
|
|
255
|
+
statistic?: 'mean' | 'median';
|
|
256
|
+
}
|
|
257
|
+
/**
|
|
258
|
+
* The Evidence Bus. For each objective, pair candidate vs baseline by full
|
|
259
|
+
* cellId and bootstrap a CI on the good-direction paired delta. Reuses the
|
|
260
|
+
* exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
|
|
261
|
+
* a single source of truth governs pairing granularity + scale handling.
|
|
262
|
+
*/
|
|
263
|
+
declare function buildEvidenceVector<TArtifact, TScenario extends Scenario>(ctx: GateContext<TArtifact, TScenario>, objectives: PromotionObjective[], opts?: BuildEvidenceVectorOptions): EvidenceVector;
|
|
264
|
+
/**
|
|
265
|
+
* The default strategy: symmetric multi-objective Pareto significance. Ship iff
|
|
266
|
+
* the candidate weakly dominates the baseline at the confidence level — no axis
|
|
267
|
+
* credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
|
|
268
|
+
* (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
|
|
269
|
+
* need_more_work. Statistically equivalent → hold (never ship noise).
|
|
270
|
+
*/
|
|
271
|
+
declare const paretoPolicy: PromotionPolicy;
|
|
272
|
+
interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
|
|
273
|
+
/** The objective vector. Every axis is both a gain source and a safety floor. */
|
|
274
|
+
objectives: PromotionObjective[];
|
|
275
|
+
/** Strategy applied to the evidence vector. Default `paretoPolicy`. Override
|
|
276
|
+
* to run a stricter/looser strategy over the SAME bus (competing policies). */
|
|
277
|
+
policy?: PromotionPolicy;
|
|
278
|
+
/** Override the gate name in reports. */
|
|
279
|
+
name?: string;
|
|
280
|
+
}
|
|
281
|
+
/**
|
|
282
|
+
* Wrap the bus + a policy as a `Gate`. Plugs into the existing
|
|
283
|
+
* `runImprovementLoop({ gate })` slot and composes via `composeGate`; default
|
|
284
|
+
* loop behavior is unchanged because consumers opt in by passing this gate.
|
|
285
|
+
*/
|
|
286
|
+
declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: ParetoSignificanceGateOptions): Gate<TArtifact, TScenario>;
|
|
287
|
+
//#endregion
|
|
288
|
+
export { powerPreflight as S, paretoFrontier as _, ObjectiveSource as a, PowerPreflight as b, PromotionPolicy as c, paretoSignificanceGate as d, Direction as f, dominates as g, crowdingDistance as h, EvidenceVector as i, buildEvidenceVector as l, ParetoResult as m, AxisVerdict as n, ParetoSignificanceGateOptions as o, Objective as p, BuildEvidenceVectorOptions as r, PromotionObjective as s, AxisEvidence as t, paretoPolicy as u, paretoFrontierWithCrowding as v, PowerPreflightOptions as x, scalarScore as y };
|
|
289
|
+
//# sourceMappingURL=promotion-policy-ChWhTDBH.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"promotion-policy-ChWhTDBH.d.ts","names":[],"sources":["../src/campaign/gates/power-preflight.ts","../src/pareto.ts","../src/campaign/gates/promotion-policy.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;UAwBiB;;EAEf;;;EAGA;;EAEA;;EAEA;;;;;;;EAOA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA;;;EAGA;EACA;EACA;;;EAGA;;EAEA;;;;;;iBAec,eAAe,MAAM,wBAAwB;;;;;;;;;;;;;;;;;;KClEjD;UAEK,UAAU;;EAEzB;EACA,WAAW;EACX,QAAQ,WAAW;;UAGJ,aAAa;EAC5B,UAAU;EACV,WAAW;;EAEX,cAAc;IAAQ,WAAW;IAAG,WAAW;;;;iBAIjC,UAAU,GAAG,GAAG,GAAG,GAAG,GAAG,YAAY,UAAU;;;;;;iBAmB/C,eAAe,GAAG,YAAY,KAAK,YAAY,UAAU,OAAO,aAAa;;;;;;;;;;iBA4B7E,YAAY,GAC1B,YAAY,KACZ,YAAY,UAAU,MACtB;EAAW,UAAU,QAAQ;IAC5B;EAAQ,WAAW;EAAG;;;;;;;;;;;;;;iBAyCT,iBAAiB,GAC/B,YAAY,KACZ,YAAY,UAAU,OACrB;EAAQ,WAAW;EAAG;;;;;;;iBA6BT,2BAA2B,GACzC,YAAY,KACZ,YAAY,UAAU,OACrB;EAAQ,WAAW;EAAG;;;;;;KCtHb;EAAoB;;EAAwB;EAAmB;;UAE1D;;EAEf;EACA,QAAQ;;;EAGR,WAAW;;;;EAIX;;;;;EAKA;;;KAIU;UAEK;EACf;EACA,QAAQ;EACR,WAAW;;;;;;;;EAQX,WAAW;;;;;EAKX;;;EAGA;IAAM;IAAa;;;EAEnB,mBAAmB;;EAEnB,SAAS;;;EAGT;;EAEA;EACA;EACA,gBAAgB;EAChB;EACA;EACA,SAAS;;UAGM;;EAEf,MAAM;;;EAGN;;;EAGA;IAAQ;IAAmB;;;;;;KAMjB,mBAAmB,IAAI,mBAAmB;UAErC;;;;EAIf;;EAEA;;EAEA;;EAEA;;;EAGA;;;;;;;;iBASc,oBAAoB,WAAW,kBAAkB,UAC/D,KAAK,YAAY,WAAW,YAC5B,YAAY,sBACZ,OAAM,6BACL;;;;;;;;cAwIU,cAAc;UAiFV,sCAAsC;;EAErD,YAAY;;;EAGZ,SAAS;;EAET;;;;;;;iBAQc,uBAAuB,qBAAqB,kBAAkB,WAAW,UACvF,SAAS,gCACR,KAAK,WAAW"}
|