@tangle-network/agent-eval 0.144.5 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +471 -89
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +8 -6
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
- package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
- package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +11 -10
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
- package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
- package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
- package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
- package/dist/errors-D-LKuDhb.js.map +1 -1
- package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
- package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
- package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
- package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
- package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +252 -451
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +271 -751
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
- package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
- package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
- package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +2 -2
- package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
- package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +19 -9
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
- package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
- package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
- package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
- package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
- package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
- package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
- package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/building-doctrine.md +15 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +21 -4
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
- package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
- package/dist/campaign-ClpnD7Ug.js.map +0 -1
- package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
- package/dist/index-CpxZSlB7.d.ts.map +0 -1
- package/dist/index-CsuAo2-J.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-D3EoChAU.js.map +0 -1
- package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
- package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
- package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
- package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
- package/dist/types-BjMFz88h.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -1,15 +1,16 @@
|
|
|
1
|
-
import { l as ValidationError, t as AgentEvalError } from "./errors-
|
|
2
|
-
import { T as RunPaidCallInput, m as CostReceipt, p as CostProvenance } from "./cost-ledger-
|
|
3
|
-
import { a as RunRecord, s as RunSplitTag } from "./run-record-
|
|
4
|
-
import { f as AnalystUsageReceipt, i as AnalystFinding, l as AnalystRunResult,
|
|
5
|
-
import { n as CompletionVerdict, o as ProducedState, r as CorrectnessChecker, t as CompletionRequirement } from "./completion-verifier-
|
|
6
|
-
import {
|
|
7
|
-
import { A as LabeledScenarioWrite, D as LabeledScenarioSampleArgs, E as LabeledScenarioRecord, O as LabeledScenarioSource, R as Scenario, S as JudgeConfig, T as LabelTrust, a as CampaignResult,
|
|
8
|
-
import {
|
|
9
|
-
import {
|
|
10
|
-
import {
|
|
11
|
-
import
|
|
12
|
-
import {
|
|
1
|
+
import { l as ValidationError, t as AgentEvalError } from "./errors-CKPfb2aH.js";
|
|
2
|
+
import { T as RunPaidCallInput, m as CostReceipt, p as CostProvenance } from "./cost-ledger-Bv_e8XHY.js";
|
|
3
|
+
import { a as RunRecord, s as RunSplitTag } from "./run-record-DdSa93_W.js";
|
|
4
|
+
import { f as AnalystUsageReceipt, i as AnalystFinding, l as AnalystRunResult, w as TraceAnalysisStore } from "./types-DF_Udrp-.js";
|
|
5
|
+
import { n as CompletionVerdict, o as ProducedState, r as CorrectnessChecker, t as CompletionRequirement } from "./completion-verifier-foUCLif_.js";
|
|
6
|
+
import { D as AnalystIssueExpectation, d as AnalystBenchmarkCase, h as AnalystBenchmarkLabelState, t as AgentProfile$1 } from "./agent-profile-CgDTo40f.js";
|
|
7
|
+
import { A as LabeledScenarioWrite, D as LabeledScenarioSampleArgs, E as LabeledScenarioRecord, O as LabeledScenarioSource, R as Scenario, S as JudgeConfig, T as LabelTrust, a as CampaignResult, d as DispatchContext, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, p as Gate, u as ComponentSurface } from "./types-DYuNHo9R.js";
|
|
8
|
+
import { y as PairedBootstrapResult } from "./statistics-D6Uebe_4.js";
|
|
9
|
+
import { Ht as CampaignStorage, It as PlanCampaignRunOptions, Pt as CampaignRunPlan } from "./skillopt-optimization-method-BO7NAl3b.js";
|
|
10
|
+
import { N as PairedArmsComparison } from "./statistical-heldout-Dn9ruizm.js";
|
|
11
|
+
import "./promotion-policy-ChWhTDBH.js";
|
|
12
|
+
import { D as LedgerHash, m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-CvSN3IG1.js";
|
|
13
|
+
import { AgentProfile } from "@tangle-network/agent-interface";
|
|
13
14
|
//#region src/campaign/analyst-surface.d.ts
|
|
14
15
|
interface TraceAnalystScenario extends Scenario {
|
|
15
16
|
kind: 'trace-analyst';
|
|
@@ -34,152 +35,6 @@ interface BuildTraceAnalystSurfaceDispatchOptions {
|
|
|
34
35
|
declare function buildTraceAnalystSurfaceDispatch(options: BuildTraceAnalystSurfaceDispatchOptions): (surface: MutableSurface, scenario: TraceAnalystScenario, context: DispatchContext) => Promise<TraceAnalystArtifact>;
|
|
35
36
|
declare function traceAnalystQualityJudge(): JudgeConfig<TraceAnalystArtifact, TraceAnalystScenario>;
|
|
36
37
|
//#endregion
|
|
37
|
-
//#region src/paired-arms.d.ts
|
|
38
|
-
/** One arm observation of one work item. Structural on purpose: callers
|
|
39
|
-
* project their own record type (e.g. a `RunRecord`) into this shape. */
|
|
40
|
-
interface PairedArmRow {
|
|
41
|
-
/** Matching key — rows sharing a `pairKey` across both arms form pairs
|
|
42
|
-
* (typically the task/scenario/seed identity). */
|
|
43
|
-
pairKey: string;
|
|
44
|
-
/** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on
|
|
45
|
-
* every row of a `pairKey` that has more than one rep in either arm; reps
|
|
46
|
-
* then pair only on exact (`pairKey`, `repKey`) match, never on outcome
|
|
47
|
-
* content. Optional when each arm has at most one rep of the item. */
|
|
48
|
-
repKey?: string;
|
|
49
|
-
/** Arm label this row was produced under. */
|
|
50
|
-
arm: string;
|
|
51
|
-
/** Binary outcome; omit when the comparison has no pass/fail notion. */
|
|
52
|
-
pass?: boolean;
|
|
53
|
-
/** Named numeric measurements (score, cost, latency, …). */
|
|
54
|
-
metrics?: Record<string, number>;
|
|
55
|
-
}
|
|
56
|
-
interface PairArmsOptions {
|
|
57
|
-
/** Arm treated as the control side of every pair. */
|
|
58
|
-
baselineArm: string;
|
|
59
|
-
/** Arm treated as the treatment side of every pair. */
|
|
60
|
-
treatmentArm: string;
|
|
61
|
-
}
|
|
62
|
-
/** One matched (baseline, treatment) observation of the same work item. */
|
|
63
|
-
interface MatchedPair {
|
|
64
|
-
pairKey: string;
|
|
65
|
-
/** 0-based position of this pair within its `pairKey`, ordered by sorted
|
|
66
|
-
* `repKey` (always 0 for a single-rep item). The rep identity itself is on
|
|
67
|
-
* the rows (`baseline.repKey` / `treatment.repKey`). */
|
|
68
|
-
repIndex: number;
|
|
69
|
-
baseline: PairedArmRow;
|
|
70
|
-
treatment: PairedArmRow;
|
|
71
|
-
}
|
|
72
|
-
interface PairArmsResult {
|
|
73
|
-
/** Matched pairs, ordered by (`pairKey`, `repIndex`). */
|
|
74
|
-
pairs: MatchedPair[];
|
|
75
|
-
/** Baseline rows left without a treatment counterpart — reported, never
|
|
76
|
-
* silently dropped. */
|
|
77
|
-
unpairedBaseline: PairedArmRow[];
|
|
78
|
-
/** Treatment rows left without a baseline counterpart. */
|
|
79
|
-
unpairedTreatment: PairedArmRow[];
|
|
80
|
-
}
|
|
81
|
-
/**
|
|
82
|
-
* Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
|
|
83
|
-
*
|
|
84
|
-
* A `pairKey` with at most one row per arm pairs directly, no `repKey`
|
|
85
|
-
* needed. A `pairKey` with multiple reps in either arm requires `repKey` on
|
|
86
|
-
* every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
|
|
87
|
-
* match — pairing is keyed purely on row identity, never on outcome content
|
|
88
|
-
* (outcome-keyed matching deflates discordant counts and biases McNemar), and
|
|
89
|
-
* is therefore independent of input order. Reps whose `repKey` has no
|
|
90
|
-
* counterpart in the other arm, and items present in only one arm, land in
|
|
91
|
-
* the unpaired lists — reported, never truncated.
|
|
92
|
-
*
|
|
93
|
-
* Fail-loud: throws when either named arm has zero rows (an unknown arm
|
|
94
|
-
* name would otherwise read as "everything unpaired"), when the two arm
|
|
95
|
-
* names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
|
|
96
|
-
* when a (`pairKey`, arm) group repeats a `repKey` (the match would be
|
|
97
|
-
* ambiguous).
|
|
98
|
-
*/
|
|
99
|
-
declare function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult;
|
|
100
|
-
/** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */
|
|
101
|
-
interface PairedCorrectness {
|
|
102
|
-
/** Discordant pairs where the treatment passed and the baseline failed. */
|
|
103
|
-
b10: number;
|
|
104
|
-
/** Discordant pairs where the baseline passed and the treatment failed. */
|
|
105
|
-
b01: number;
|
|
106
|
-
/** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */
|
|
107
|
-
mcnemar: McNemarResult;
|
|
108
|
-
/** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */
|
|
109
|
-
riskDifference: RiskDifferenceResult;
|
|
110
|
-
}
|
|
111
|
-
/** Paired delta summary for one named metric (delta = treatment − baseline). */
|
|
112
|
-
interface PairedMetricDelta {
|
|
113
|
-
name: string;
|
|
114
|
-
/** Pairs where BOTH sides carry a finite value for this metric. */
|
|
115
|
-
n: number;
|
|
116
|
-
/** Pairs where at least one side does not carry the metric. */
|
|
117
|
-
nMissing: number;
|
|
118
|
-
/** Median paired delta, or null when `n === 0`. */
|
|
119
|
-
medianDelta: number | null;
|
|
120
|
-
/** Mean paired delta, or null when `n === 0`. */
|
|
121
|
-
meanDelta: number | null;
|
|
122
|
-
/** Bootstrap CI on the paired delta (`pairedBootstrap`); null when
|
|
123
|
-
* `n === 0` — a zero-width [0, 0] interval on no data would read as a
|
|
124
|
-
* measured tight null. */
|
|
125
|
-
bootstrapCi: PairedBootstrapResult | null;
|
|
126
|
-
/** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */
|
|
127
|
-
wilcoxon: {
|
|
128
|
-
w: number;
|
|
129
|
-
p: number;
|
|
130
|
-
} | null;
|
|
131
|
-
}
|
|
132
|
-
interface ComparePairedArmsOptions extends PairArmsOptions {
|
|
133
|
-
/** Metrics to compare. Default: every metric name observed on any matched
|
|
134
|
-
* pair, sorted. A name that appears on no pair is still reported (with
|
|
135
|
-
* `n = 0`) so a misspelled metric is visible instead of vanishing. */
|
|
136
|
-
metricNames?: string[];
|
|
137
|
-
/** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */
|
|
138
|
-
bootstrap?: PairedBootstrapOptions;
|
|
139
|
-
}
|
|
140
|
-
interface PairedArmsComparison {
|
|
141
|
-
nPairs: number;
|
|
142
|
-
nUnpairedBaseline: number;
|
|
143
|
-
nUnpairedTreatment: number;
|
|
144
|
-
/** null when no matched pair carries `pass` on both sides — a pass/fail
|
|
145
|
-
* verdict over rows that never measured pass/fail would be fabricated. */
|
|
146
|
-
correctness: PairedCorrectness | null;
|
|
147
|
-
metricDeltas: PairedMetricDelta[];
|
|
148
|
-
}
|
|
149
|
-
/**
|
|
150
|
-
* Full matched-pair arm comparison: pair via {@link pairArms}, then compose
|
|
151
|
-
* the paired estimators from `statistics` over the matched pairs.
|
|
152
|
-
*
|
|
153
|
-
* Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
|
|
154
|
-
* is that subset's size); each metric uses only the pairs where both sides
|
|
155
|
-
* carry a finite value for it, with the remainder counted in `nMissing`.
|
|
156
|
-
* Deltas are treatment − baseline throughout.
|
|
157
|
-
*
|
|
158
|
-
* Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
|
|
159
|
-
* non-finite metric value — silently treating corrupt telemetry as "metric
|
|
160
|
-
* absent" would misreport it as missing coverage.
|
|
161
|
-
*/
|
|
162
|
-
declare function comparePairedArms(rows: readonly PairedArmRow[], opts: ComparePairedArmsOptions): PairedArmsComparison;
|
|
163
|
-
interface MatchedRunRecordPair {
|
|
164
|
-
pairKey: string;
|
|
165
|
-
repKey: string;
|
|
166
|
-
baseline: RunRecord;
|
|
167
|
-
treatment: RunRecord;
|
|
168
|
-
}
|
|
169
|
-
interface PairRunRecordsResult {
|
|
170
|
-
pairs: MatchedRunRecordPair[];
|
|
171
|
-
unpairedBaseline: RunRecord[];
|
|
172
|
-
unpairedTreatment: RunRecord[];
|
|
173
|
-
}
|
|
174
|
-
/**
|
|
175
|
-
* Pair two RunRecord arms by the identity of the evaluated work:
|
|
176
|
-
* `(experimentId, scenarioId, seed)`.
|
|
177
|
-
*
|
|
178
|
-
* Falling back to array order, candidate id, or experiment id can compare
|
|
179
|
-
* different tasks and fabricate lift. Duplicate identities throw.
|
|
180
|
-
*/
|
|
181
|
-
declare function pairRunRecords(baselineRuns: readonly RunRecord[], treatmentRuns: readonly RunRecord[]): PairRunRecordsResult;
|
|
182
|
-
//#endregion
|
|
183
38
|
//#region src/campaign/cross-surface-types.d.ts
|
|
184
39
|
/** Whether one candidate attempt produced a usable executable outcome. */
|
|
185
40
|
type CrossSurfaceAttemptCompleteness = 'complete' | 'missing' | 'invalid';
|
|
@@ -525,409 +380,6 @@ interface NeutralizationGateOptions<TScenario extends Scenario = Scenario> {
|
|
|
525
380
|
*/
|
|
526
381
|
declare function neutralizationGate<TArtifact, TScenario extends Scenario>(options: NeutralizationGateOptions<TScenario>): Gate<TArtifact, TScenario>;
|
|
527
382
|
//#endregion
|
|
528
|
-
//#region src/pre-registration.d.ts
|
|
529
|
-
/**
|
|
530
|
-
* Pre-registered hypotheses — declare what you're testing BEFORE the
|
|
531
|
-
* run, check it AFTER. Prevents p-hacking, optional stopping, and the
|
|
532
|
-
* "we ran until it looked good" failure mode.
|
|
533
|
-
*
|
|
534
|
-
* Manifest is a plain JSON-friendly object. Sign it with a content hash
|
|
535
|
-
* + timestamp; the registered record becomes immutable. Post-run,
|
|
536
|
-
* evaluate the manifest against observed results — the library refuses
|
|
537
|
-
* to let you re-interpret a different metric as the declared one.
|
|
538
|
-
*/
|
|
539
|
-
interface HypothesisManifest {
|
|
540
|
-
id: string;
|
|
541
|
-
/** Human prose — goes into the audit trail. */
|
|
542
|
-
hypothesis: string;
|
|
543
|
-
/** Metric the hypothesis claims to move. */
|
|
544
|
-
metric: string;
|
|
545
|
-
/** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
|
|
546
|
-
direction: 'increase' | 'decrease';
|
|
547
|
-
/** Minimum effect size to count (same units as the metric). */
|
|
548
|
-
minEffect: number;
|
|
549
|
-
/** Alpha threshold. */
|
|
550
|
-
alpha: number;
|
|
551
|
-
/** Target statistical power at which sample size was pre-computed. */
|
|
552
|
-
power: number;
|
|
553
|
-
/** Declared N per arm before running. */
|
|
554
|
-
preRegisteredN: number;
|
|
555
|
-
/** ISO8601 timestamp the manifest was registered. */
|
|
556
|
-
registeredAt: string;
|
|
557
|
-
/** Optional identifiers to tie into the trace corpus. */
|
|
558
|
-
baselineLabel?: string;
|
|
559
|
-
candidateLabel?: string;
|
|
560
|
-
}
|
|
561
|
-
/**
|
|
562
|
-
* Identifier for the hashing scheme used to produce `contentHash`.
|
|
563
|
-
*
|
|
564
|
-
* `'sha256-content'` — sha256 hex over the canonicalized manifest with
|
|
565
|
-
* the `contentHash` and `algo` fields stripped. Held as a string union
|
|
566
|
-
* so future schemes can be added without breaking parsers; SignedManifest
|
|
567
|
-
* values without `algo` deserialize cleanly because the field is optional.
|
|
568
|
-
*/
|
|
569
|
-
type SignedManifestAlgo = 'sha256-content';
|
|
570
|
-
interface SignedManifest extends HypothesisManifest {
|
|
571
|
-
/** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
|
|
572
|
-
contentHash: string;
|
|
573
|
-
/**
|
|
574
|
-
* Algorithm string describing how `contentHash` was produced.
|
|
575
|
-
*
|
|
576
|
-
* Optional on the type so serialized manifests without it still parse,
|
|
577
|
-
* but ALWAYS populated by {@link signManifest}. Consumers that want to
|
|
578
|
-
* enforce a known algorithm should reject manifests where this field
|
|
579
|
-
* is missing or unrecognized.
|
|
580
|
-
*/
|
|
581
|
-
algo?: SignedManifestAlgo;
|
|
582
|
-
}
|
|
583
|
-
interface HypothesisResult {
|
|
584
|
-
manifest: SignedManifest;
|
|
585
|
-
observedN: number;
|
|
586
|
-
observedEffect: number;
|
|
587
|
-
observedPValue: number;
|
|
588
|
-
/** True iff the observed effect hits the pre-declared direction with
|
|
589
|
-
* magnitude ≥ minEffect AND p < alpha. */
|
|
590
|
-
confirmed: boolean;
|
|
591
|
-
/** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
|
|
592
|
-
rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
|
|
593
|
-
notes?: string;
|
|
594
|
-
}
|
|
595
|
-
/**
|
|
596
|
-
* Deterministic JSON canonicalization — sort object keys recursively.
|
|
597
|
-
*
|
|
598
|
-
* Two semantically-equal objects produce byte-identical canonicalized output;
|
|
599
|
-
* this is what makes a content-hash stable across encoders, key insertion
|
|
600
|
-
* orders, and runtime versions. Exported for any consumer that needs the same
|
|
601
|
-
* canonicalization guarantee outside the manifest-signing path (e.g., signing
|
|
602
|
-
* an artifact bundle, hashing a dataset version, etc.).
|
|
603
|
-
*/
|
|
604
|
-
declare function canonicalize(v: unknown): unknown;
|
|
605
|
-
/**
|
|
606
|
-
* SHA-256 hex (full 64 chars) over the canonicalized JSON encoding of `obj`.
|
|
607
|
-
*
|
|
608
|
-
* The same primitive `signManifest` and `verifyManifest` are built on, exposed
|
|
609
|
-
* directly so consumers signing arbitrary structured content (artifact bundles,
|
|
610
|
-
* production packets, dataset manifests, etc.) don't have to re-derive
|
|
611
|
-
* canonicalize+sha256 from scratch.
|
|
612
|
-
*
|
|
613
|
-
* Stable across:
|
|
614
|
-
* - object key insertion order (canonicalization sorts keys recursively)
|
|
615
|
-
* - encoder choice (UTF-8 via TextEncoder, fixed)
|
|
616
|
-
* - runtime (uses the Web Crypto subtle digest, present in Node ≥18 and browsers)
|
|
617
|
-
*
|
|
618
|
-
* Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
|
|
619
|
-
* which takes a string input and returns a truncated 12-char prompt id.
|
|
620
|
-
* Use `hashJson` when you mean "canonicalize then hash."
|
|
621
|
-
*
|
|
622
|
-
* @example
|
|
623
|
-
* const hash = await hashJson({ id: '1', kind: 'spec' })
|
|
624
|
-
* // 'a3f1...' (64 hex chars)
|
|
625
|
-
*/
|
|
626
|
-
declare function hashJson<T>(obj: T): Promise<string>;
|
|
627
|
-
/**
|
|
628
|
-
* Sign a manifest with a SHA-256 content hash.
|
|
629
|
-
*
|
|
630
|
-
* The hash covers the canonicalized manifest with the `contentHash`
|
|
631
|
-
* and `algo` fields stripped; this lets verifiers re-sign the rest and
|
|
632
|
-
* compare. Returned manifest always carries `algo: 'sha256-content'`
|
|
633
|
-
* so downstream consumers can identify the scheme; manifests without
|
|
634
|
-
* `algo` still verify because it is stripped before hashing on both sides.
|
|
635
|
-
*/
|
|
636
|
-
declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
|
|
637
|
-
/**
|
|
638
|
-
* Verify that a signed manifest has not been tampered with.
|
|
639
|
-
*
|
|
640
|
-
* Strips `contentHash` and `algo` before re-signing so manifests without
|
|
641
|
-
* `algo` verify identically to ones that carry it.
|
|
642
|
-
*/
|
|
643
|
-
declare function verifyManifest(m: SignedManifest): Promise<boolean>;
|
|
644
|
-
/**
|
|
645
|
-
* Evaluate a pre-registered hypothesis against observed results.
|
|
646
|
-
* Mechanical — no re-interpretation permitted.
|
|
647
|
-
*/
|
|
648
|
-
declare function evaluateHypothesis(manifest: SignedManifest, observed: {
|
|
649
|
-
n: number;
|
|
650
|
-
effect: number;
|
|
651
|
-
pValue: number;
|
|
652
|
-
}): Promise<HypothesisResult>;
|
|
653
|
-
//#endregion
|
|
654
|
-
//#region src/campaign/gates/sequential.d.ts
|
|
655
|
-
type SequentialDecision = 'promote' | 'continue' | 'undecided-at-maxN';
|
|
656
|
-
interface SequentialObservation {
|
|
657
|
-
decision: SequentialDecision;
|
|
658
|
-
/** Current e-value (the betting wealth) against H0. */
|
|
659
|
-
eValue: number;
|
|
660
|
-
/** Paired deltas consumed so far. */
|
|
661
|
-
n: number;
|
|
662
|
-
/** Names the decision basis. For 'undecided-at-maxN' it states explicitly
|
|
663
|
-
* that exhausting the budget is NOT evidence of no effect. */
|
|
664
|
-
reason: string;
|
|
665
|
-
}
|
|
666
|
-
interface SequentialPairedGateOptions {
|
|
667
|
-
/** Type-I budget. With `preRegistration` bound this MUST match
|
|
668
|
-
* `manifest.alpha` (conflict throws). Default 0.05. */
|
|
669
|
-
alpha?: number;
|
|
670
|
-
/** Minimum paired deltas before a promote may fire. The stopping rule is
|
|
671
|
-
* "first n ≥ minN with e-value ≥ 1/alpha" — still a valid stopping time.
|
|
672
|
-
* Default 5. */
|
|
673
|
-
minN?: number;
|
|
674
|
-
/** Pre-registered observation budget. Required unless `preRegistration`
|
|
675
|
-
* supplies it via `preRegisteredN` (conflict throws). */
|
|
676
|
-
maxN?: number;
|
|
677
|
-
/** Bet truncation forwarded to `eProcess`. Default 0.5. */
|
|
678
|
-
maxBet?: number;
|
|
679
|
-
/** Bound on |delta| in the judge's native scale; deltas are mapped to
|
|
680
|
-
* x = (d/scale + 1)/2 ∈ [0,1]. A delta outside ±scale throws (use
|
|
681
|
-
* `detectScale` to pick 1 vs 100 BEFORE streaming). Default 1. */
|
|
682
|
-
scale?: number;
|
|
683
|
-
/** Seed for the data-independent shuffle of paired deltas in `decide(ctx)`
|
|
684
|
-
* (exchangeability guard). Default 1337. */
|
|
685
|
-
shuffleSeed?: number;
|
|
686
|
-
/** Bind the pre-registered hypothesis. Verified (content hash) at
|
|
687
|
-
* construction; alpha/maxN/direction/minEffect come FROM the manifest. */
|
|
688
|
-
preRegistration?: SignedManifest;
|
|
689
|
-
/** Override the gate name in reports. */
|
|
690
|
-
name?: string;
|
|
691
|
-
}
|
|
692
|
-
interface SequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario> extends Gate<TArtifact, TScenario> {
|
|
693
|
-
/** Streaming entry point: feed one paired per-scenario delta
|
|
694
|
-
* (candidate − baseline, native scale). Each gate instance carries ONE
|
|
695
|
-
* observe-stream; `decide(ctx)` runs on its own fresh stream and never
|
|
696
|
-
* consumes or advances this one. 'promote' is sticky; observing past the
|
|
697
|
-
* pre-registered maxN throws (extending a finished stream after seeing
|
|
698
|
-
* the result reopens optional stopping — start a NEW pre-registered
|
|
699
|
-
* test). */
|
|
700
|
-
observe(delta: number): SequentialObservation;
|
|
701
|
-
/** Read-only snapshot of the observe-stream. */
|
|
702
|
-
state(): EProcessState & {
|
|
703
|
-
decision: SequentialDecision;
|
|
704
|
-
};
|
|
705
|
-
}
|
|
706
|
-
/**
|
|
707
|
-
* Anytime-valid sequential paired gate. Conforms to the existing `Gate`
|
|
708
|
-
* contract (`decide(ctx)` consumes candidate vs baseline judge scores via
|
|
709
|
-
* `pairHoldout` — same pairing granularity as the fixed-n gates: full cellId,
|
|
710
|
-
* never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
|
|
711
|
-
* that score cells incrementally and want to stop mid-stream.
|
|
712
|
-
*
|
|
713
|
-
* Decision mapping onto the substrate's five-valued `GateDecision`:
|
|
714
|
-
* - 'promote' → 'ship'
|
|
715
|
-
* - 'continue' → 'need_more_work' (stream ended before maxN with
|
|
716
|
-
* the e-value undecided — more reps could decide)
|
|
717
|
-
* - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
|
|
718
|
-
* evidence of no effect (never a silent default)
|
|
719
|
-
*/
|
|
720
|
-
declare function sequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: SequentialPairedGateOptions): SequentialPairedGate<TArtifact, TScenario>;
|
|
721
|
-
interface SequentialDecideOptions {
|
|
722
|
-
/** Type-I budget for the early-stop evidence. Default 0.05. */
|
|
723
|
-
alpha?: number;
|
|
724
|
-
/** Minimum paired deltas before a stop may fire. Default 5. */
|
|
725
|
-
minN?: number;
|
|
726
|
-
/** Bet truncation forwarded to `eProcess`. Default 0.5. */
|
|
727
|
-
maxBet?: number;
|
|
728
|
-
/** Bound on |per-scenario composite delta|. Default 1. */
|
|
729
|
-
scale?: number;
|
|
730
|
-
}
|
|
731
|
-
interface SequentialDecideFn {
|
|
732
|
-
(args: {
|
|
733
|
-
history: GenerationRecord[];
|
|
734
|
-
}): {
|
|
735
|
-
stop: boolean;
|
|
736
|
-
reason?: string;
|
|
737
|
-
};
|
|
738
|
-
/** Read-only snapshot of the accumulated e-process (observability + tests). */
|
|
739
|
-
state(): EProcessState;
|
|
740
|
-
}
|
|
741
|
-
/**
|
|
742
|
-
* `SurfaceProposer.decide` adapter — stops the optimization loop the moment
|
|
743
|
-
* the e-process decides the loop has produced a real improvement, instead of
|
|
744
|
-
* always running `maxGenerations`.
|
|
745
|
-
*
|
|
746
|
-
* Stream: for each generation g ≥ 1, the per-scenario composite deltas of
|
|
747
|
-
* generation g's top candidate vs the generation-0 top candidate (the
|
|
748
|
-
* incumbent the loop set out to beat), paired by scenarioId. H0: no proposed
|
|
749
|
-
* surface improves any scenario's expected composite over the incumbent —
|
|
750
|
-
* under it every delta has conditional mean ≤ 0 and the e-process is valid.
|
|
751
|
-
* Once wealth ≥ 1/alpha the loop stops and hands the winner to the promotion
|
|
752
|
-
* gate (which re-scores on HELD-OUT data — this adapter only spends the
|
|
753
|
-
* exploration budget, it never promotes).
|
|
754
|
-
*
|
|
755
|
-
* Honesty caveats: (1) the incumbent's scores are measured once and shared
|
|
756
|
-
* across all generations' deltas, so type-I control is exact only insofar as
|
|
757
|
-
* those scores approximate the incumbent's true per-scenario means (more reps
|
|
758
|
-
* → tighter); (2) an UNDECIDED process never stops the loop — absence of a
|
|
759
|
-
* crossing is NOT evidence of no effect, so the loop simply runs its normal
|
|
760
|
-
* course. Calling the adapter repeatedly with a growing history consumes each
|
|
761
|
-
* generation exactly once (re-feeding an already-seen record would double-count
|
|
762
|
-
* evidence).
|
|
763
|
-
*/
|
|
764
|
-
declare function sequentialDecide(options?: SequentialDecideOptions): SequentialDecideFn;
|
|
765
|
-
//#endregion
|
|
766
|
-
//#region src/campaign/gates/statistical-heldout.d.ts
|
|
767
|
-
interface PairedHoldout {
|
|
768
|
-
/** Baseline scalar per paired cell (same order as `after`/`cellIds`). */
|
|
769
|
-
before: number[];
|
|
770
|
-
/** Candidate scalar per paired cell. */
|
|
771
|
-
after: number[];
|
|
772
|
-
/** The full cellIds (`scenario:rep`) that paired, in order. */
|
|
773
|
-
cellIds: string[];
|
|
774
|
-
}
|
|
775
|
-
/**
|
|
776
|
-
* Pair candidate vs baseline holdout observations by FULL cellId. `select`
|
|
777
|
-
* pulls the scalar from a cell's judge reports (composite, or a named
|
|
778
|
-
* dimension); a cell contributes the mean of `select` across its judges. Cells
|
|
779
|
-
* whose scenario is not in `scenarioIds`, or where `select` is undefined for
|
|
780
|
-
* every judge on either side, are skipped on BOTH sides so the arrays stay
|
|
781
|
-
* paired. Throws when the two maps disagree on which holdout cells exist — a
|
|
782
|
-
* load-bearing invariant: the baseline + winner holdout campaigns run the same
|
|
783
|
-
* scenarios with the same seed base, so their cellIds MUST align; a mismatch
|
|
784
|
-
* means a silent pairing bug, not a soft fallback.
|
|
785
|
-
*/
|
|
786
|
-
declare function pairHoldout(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, select: (s: JudgeScore) => number | undefined): PairedHoldout;
|
|
787
|
-
interface HeldoutSignificance {
|
|
788
|
-
paired: PairedHoldout;
|
|
789
|
-
/**
|
|
790
|
-
* The paired bootstrap on the requested statistic (MEAN by default — see the
|
|
791
|
-
* tie note on `heldoutSignificance`).
|
|
792
|
-
*
|
|
793
|
-
* DIAGNOSTIC, not necessarily the interval the verdict keyed on. On a
|
|
794
|
-
* two-point (pass/fail) outcome the decision routes to Tango's score interval
|
|
795
|
-
* instead, because a percentile bootstrap of the mean over a three-atom
|
|
796
|
-
* lattice is not a valid interval at a nonzero margin. Read
|
|
797
|
-
* `decision.low`/`decision.high` for the interval that actually decided, and
|
|
798
|
-
* `decisionStatistic` for which one it is.
|
|
799
|
-
*/
|
|
800
|
-
bootstrap: PairedBootstrapResult;
|
|
801
|
-
/** The MEDIAN paired-delta bootstrap, reported as a diagnostic. When many
|
|
802
|
-
* scenarios are tied (both sides solve them), the median is pinned near 0
|
|
803
|
-
* regardless of the mean lift — comparing the two exposes tie-domination. */
|
|
804
|
-
medianBootstrap: PairedBootstrapResult;
|
|
805
|
-
/**
|
|
806
|
-
* The full promotion decision: which estimator the outcome's shape admits,
|
|
807
|
-
* the interval it produced, McNemar's exact veto on the two-point path, and
|
|
808
|
-
* whether the interval was zero-width (no evidence in either direction). The
|
|
809
|
-
* single source of `significant`.
|
|
810
|
-
*/
|
|
811
|
-
decision: PairedPromotionDecision;
|
|
812
|
-
/** Which paired estimator the verdict was decided on. */
|
|
813
|
-
decisionStatistic: PairedDecisionStatistic;
|
|
814
|
-
/** McNemar's exact evidence on the two-point path; null otherwise. */
|
|
815
|
-
mcnemar: PairedMcNemarEvidence | null;
|
|
816
|
-
/** Fraction of paired observations that are exact ties (|delta| < 1e-9). A
|
|
817
|
-
* high tie fraction is WHY a median-based gate would have missed a real lift;
|
|
818
|
-
* it is the observability the tie fix adds. */
|
|
819
|
-
tieFraction: number;
|
|
820
|
-
/** n paired observations. */
|
|
821
|
-
n: number;
|
|
822
|
-
/** Effective minimum after applying the bootstrap's hard statistical floor. */
|
|
823
|
-
minimumRequired: number;
|
|
824
|
-
/** Statistical method that carried the decision. */
|
|
825
|
-
decisionMethod: PairedDecisionMethod;
|
|
826
|
-
/** Exact one-sided p-value on the small-sample path; otherwise null. */
|
|
827
|
-
pValue: number | null;
|
|
828
|
-
/** True iff n >= minimumRequired, the DECIDING interval has nonzero width,
|
|
829
|
-
* its lower bound clears the threshold, and McNemar's exact test does not
|
|
830
|
-
* veto at a non-negative threshold. */
|
|
831
|
-
significant: boolean;
|
|
832
|
-
/** Set when n < minimumRequired — too little evidence to claim significance. */
|
|
833
|
-
fewRuns: boolean;
|
|
834
|
-
}
|
|
835
|
-
interface HeldoutSignificanceOptions {
|
|
836
|
-
deltaThreshold?: number;
|
|
837
|
-
minProductiveRuns?: number;
|
|
838
|
-
confidence?: number;
|
|
839
|
-
resamples?: number;
|
|
840
|
-
/** Fixed by default for a deterministic, reproducible gate verdict. */
|
|
841
|
-
seed?: number;
|
|
842
|
-
statistic?: 'mean' | 'median';
|
|
843
|
-
}
|
|
844
|
-
/**
|
|
845
|
-
* Significance of the held-out composite lift: ship only when the lower bound
|
|
846
|
-
* of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
|
|
847
|
-
* 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
|
|
848
|
-
* scale.
|
|
849
|
-
*
|
|
850
|
-
* The decision is delegated whole to {@link decidePairedPromotion}, the one
|
|
851
|
-
* copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
|
|
852
|
-
* also calls. That module's header carries the measurements; the short version
|
|
853
|
-
* is three guards a bare `bootstrap.low > threshold` does not have:
|
|
854
|
-
*
|
|
855
|
-
* - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
|
|
856
|
-
* only paired-binary construction that stays valid at a nonzero margin;
|
|
857
|
-
* - McNemar's exact test VETOES at any non-negative threshold;
|
|
858
|
-
* - a ZERO-WIDTH interval is refused rather than promoted, in either
|
|
859
|
-
* direction — [0,0] clears every negative threshold and [g,g] clears every
|
|
860
|
-
* threshold below g, and both are an absence of evidence, not a result.
|
|
861
|
-
*
|
|
862
|
-
* Measured on this function before those guards landed, at a nominal 5 %:
|
|
863
|
-
* 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
|
|
864
|
-
* and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
|
|
865
|
-
* delta is exactly 0.
|
|
866
|
-
*
|
|
867
|
-
* At small n, where the percentile bootstrap is descriptive only, a
|
|
868
|
-
* pre-registered exact sign test still carries the bootstrap path.
|
|
869
|
-
*/
|
|
870
|
-
declare function heldoutSignificance(paired: PairedHoldout, opts?: HeldoutSignificanceOptions): HeldoutSignificance;
|
|
871
|
-
interface DimensionRegression {
|
|
872
|
-
dimension: string;
|
|
873
|
-
/** Paired bootstrap on (candidate − baseline). DIAGNOSTIC on a pass/fail
|
|
874
|
-
* dimension, where `ci` carries the interval that decided instead. */
|
|
875
|
-
bootstrap: PairedBootstrapResult;
|
|
876
|
-
/** Which paired statistic `bootstrap.low` is the lower bound of. `'mean'`
|
|
877
|
-
* unless the caller asked for the median. `bootstrap.median` still carries
|
|
878
|
-
* the median point estimate either way. */
|
|
879
|
-
bootstrapStatistic: 'median' | 'mean';
|
|
880
|
-
/** The interval `regressed` was decided on, in the dimension's native units. */
|
|
881
|
-
ci: {
|
|
882
|
-
low: number;
|
|
883
|
-
high: number;
|
|
884
|
-
};
|
|
885
|
-
/** Which estimator produced `ci`. */
|
|
886
|
-
decisionStatistic: PairedDecisionStatistic;
|
|
887
|
-
/** McNemar's exact evidence on a pass/fail dimension; null otherwise. */
|
|
888
|
-
mcnemar: PairedMcNemarEvidence | null;
|
|
889
|
-
/** `ci` has zero width — no evidence in either direction. */
|
|
890
|
-
indeterminate: boolean;
|
|
891
|
-
/** True iff the candidate may have regressed this dimension by more than
|
|
892
|
-
* tolerance: the lower bound of the DECIDING interval on (candidate −
|
|
893
|
-
* baseline) is below −tolerance, OR the exact small-sample test proves a drop
|
|
894
|
-
* past tolerance. */
|
|
895
|
-
regressed: boolean;
|
|
896
|
-
tolerance: number;
|
|
897
|
-
n: number;
|
|
898
|
-
}
|
|
899
|
-
/** Detect the native scale of a set of scores: 0-100 when any magnitude clears
|
|
900
|
-
* 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
|
|
901
|
-
* expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
|
|
902
|
-
declare function detectScale(values: number[]): 1 | 100;
|
|
903
|
-
/** Per-critical-dimension regression guard. For each dimension, pair the
|
|
904
|
-
* candidate vs baseline values by full cellId and bootstrap the paired delta;
|
|
905
|
-
* a dimension is "regressed" when the CI lower bound < −tolerance (conservative
|
|
906
|
-
* — blocks if the credible worst case exceeds tolerance, which is the right
|
|
907
|
-
* posture for safety dimensions like `hallucination_free`). When `tolerance`
|
|
908
|
-
* is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
|
|
909
|
-
*
|
|
910
|
-
* The interval comes from {@link decidePairedPromotion}, so a pass/fail
|
|
911
|
-
* dimension is judged on Tango's score interval rather than a percentile
|
|
912
|
-
* bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
|
|
913
|
-
* is not a valid interval at one. That matters most here because this guard
|
|
914
|
-
* fails OPEN by construction: `tolerance` is positive, so an interval pinned at
|
|
915
|
-
* [0,0] never satisfies `low < −tolerance` and a real regression on a safety
|
|
916
|
-
* dimension would be reported as `regressed: false`. On the median it fails the
|
|
917
|
-
* same way for the same reason — when most pairs tie, which is automatic for a
|
|
918
|
-
* pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
|
|
919
|
-
* to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
|
|
920
|
-
* restore the pre-0.134 behaviour. */
|
|
921
|
-
declare function dimensionRegressions(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, criticalDimensions: string[], opts?: {
|
|
922
|
-
tolerance?: number;
|
|
923
|
-
confidence?: number;
|
|
924
|
-
resamples?: number;
|
|
925
|
-
seed?: number;
|
|
926
|
-
/** Paired statistic the CI is computed on. Default `'mean'` — see
|
|
927
|
-
* {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */
|
|
928
|
-
statistic?: 'mean' | 'median';
|
|
929
|
-
}): DimensionRegression[];
|
|
930
|
-
//#endregion
|
|
931
383
|
//#region src/campaign/grounded-reflection.d.ts
|
|
932
384
|
/**
|
|
933
385
|
* Evidence grounding for reflective optimizers (GEPA-style revise loops).
|
|
@@ -1118,92 +570,6 @@ type RuntimeEventLike = ToolCallEventLike | ArtifactEventLike | ProposalEventLik
|
|
|
1118
570
|
*/
|
|
1119
571
|
declare function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState;
|
|
1120
572
|
//#endregion
|
|
1121
|
-
//#region src/agent-profile.d.ts
|
|
1122
|
-
/**
|
|
1123
|
-
* The agentic coding harnesses an eval sweeps by default — the ones we care about
|
|
1124
|
-
* ranking. This is the SINGLE source of that list; consumers import it instead of
|
|
1125
|
-
* re-declaring their own (a re-declared list is how the fleet drifts). Pass an
|
|
1126
|
-
* explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
|
|
1127
|
-
* harness) to widen beyond these.
|
|
1128
|
-
*/
|
|
1129
|
-
declare const CODING_HARNESSES: readonly HarnessType[];
|
|
1130
|
-
interface ProfileAxisSpec {
|
|
1131
|
-
/** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the
|
|
1132
|
-
* harness and model vary. `model.default` is the fallback model. */
|
|
1133
|
-
base: AgentProfile;
|
|
1134
|
-
/** Harnesses to cross. Default: {@link CODING_HARNESSES}. */
|
|
1135
|
-
harnesses?: readonly HarnessType[];
|
|
1136
|
-
/** Models to cross. Default: `[base.model.default]` — one model, i.e. today's
|
|
1137
|
-
* single-model behaviour, so omitting this never changes an existing run. */
|
|
1138
|
-
models?: readonly string[];
|
|
1139
|
-
/** Force every (harness, model) pair verbatim, even ones the harness can't run —
|
|
1140
|
-
* for deliberately testing failure modes. Default (false): SNAP instead — a
|
|
1141
|
-
* vendor-locked harness runs only the swept models in its family, or its native
|
|
1142
|
-
* default when it supports none, so no harness is dropped and none gets a
|
|
1143
|
-
* guaranteed-failing foreign-model cell. */
|
|
1144
|
-
keepIncompatible?: boolean;
|
|
1145
|
-
}
|
|
1146
|
-
/** Model sentinel for a vendor-locked harness that supports none of the swept models:
|
|
1147
|
-
* it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
|
|
1148
|
-
* resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
|
|
1149
|
-
* model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
|
|
1150
|
-
* table that would rot as router catalogs change. */
|
|
1151
|
-
declare const HARNESS_NATIVE_MODEL = "default";
|
|
1152
|
-
/**
|
|
1153
|
-
* Expand a base profile across the harness × model matrix into the `AgentProfile[]`
|
|
1154
|
-
* that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
|
|
1155
|
-
* which models do we evaluate" lives, so no product hand-rolls its own harness list
|
|
1156
|
-
* or column→profile mapping (the pattern that let those copies drift and silently
|
|
1157
|
-
* break the harness pivot).
|
|
1158
|
-
*
|
|
1159
|
-
* Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,
|
|
1160
|
-
* and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both
|
|
1161
|
-
* metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row
|
|
1162
|
-
* and results join back by harness/model via {@link harnessAxisOf} with no
|
|
1163
|
-
* hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or
|
|
1164
|
-
* its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every
|
|
1165
|
-
* requested harness runs; `keepIncompatible` forces every pair verbatim.
|
|
1166
|
-
*
|
|
1167
|
-
* Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
|
|
1168
|
-
* everything we care about" switch, identical in shape whether one harness or all.
|
|
1169
|
-
*/
|
|
1170
|
-
declare function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[];
|
|
1171
|
-
/**
|
|
1172
|
-
* Read the (harness, model) a matrix cell ran under, off a profile or a result row's
|
|
1173
|
-
* profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
|
|
1174
|
-
* wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
|
|
1175
|
-
* this instead of recomputing an id (recomputing the wrong key is what broke the pivot
|
|
1176
|
-
* in the hand-rolled copies).
|
|
1177
|
-
*/
|
|
1178
|
-
declare function harnessAxisOf(profile: Pick<AgentProfile, 'metadata'>): {
|
|
1179
|
-
harness: HarnessType;
|
|
1180
|
-
model: string;
|
|
1181
|
-
} | undefined;
|
|
1182
|
-
/**
|
|
1183
|
-
* Collision-resistant, path-safe, human-readable profile id for eval artifacts.
|
|
1184
|
-
* Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
|
|
1185
|
-
* keys, and directory names where two profiles must not collapse onto one row.
|
|
1186
|
-
* The suffix is the first 64 bits of the behaviour hash, enough for ordinary
|
|
1187
|
-
* eval matrices while keeping filenames readable.
|
|
1188
|
-
*/
|
|
1189
|
-
declare function agentProfileId(profile: AgentProfile): string;
|
|
1190
|
-
/**
|
|
1191
|
-
* Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete
|
|
1192
|
-
* model id because run records reject bare/missing model aliases.
|
|
1193
|
-
*/
|
|
1194
|
-
declare function agentProfileModelId(profile: AgentProfile): string;
|
|
1195
|
-
/**
|
|
1196
|
-
* Deterministic behaviour identity for the canonical
|
|
1197
|
-
* `@tangle-network/agent-interface` AgentProfile.
|
|
1198
|
-
*
|
|
1199
|
-
* `name` and `description` are labels and do not affect the hash. Profile
|
|
1200
|
-
* `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
|
|
1201
|
-
* and extensions do affect the hash. Resource array order is hash-bearing
|
|
1202
|
-
* because mount order can change agent behaviour. Undefined fields are treated
|
|
1203
|
-
* as absent; explicit `null` fields remain hash-bearing.
|
|
1204
|
-
*/
|
|
1205
|
-
declare function agentProfileHash(profile: AgentProfile): string;
|
|
1206
|
-
//#endregion
|
|
1207
573
|
//#region src/integrity/backend-integrity.d.ts
|
|
1208
574
|
interface BackendIntegrityReport {
|
|
1209
575
|
/** Total records inspected. */
|
|
@@ -2145,5 +1511,5 @@ declare function verifyCodeSurface(surface: CodeSurface, worktreeDir?: string):
|
|
|
2145
1511
|
* identity against the checkout at `worktreeRef`. */
|
|
2146
1512
|
declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
|
|
2147
1513
|
//#endregion
|
|
2148
|
-
export { SearchPlannedEvent as $,
|
|
2149
|
-
//# sourceMappingURL=index-
|
|
1514
|
+
export { SearchPlannedEvent as $, CrossSurfacePairIncompatibilityReason as $n, ToolCallEventLike as $t, SearchAccountingAudit as A, CrossSurfaceAttemptCompleteness as An, UserStory as At, SearchCostAccounting as B, CrossSurfaceCompositionStep as Bn, RunProfileMatrixOptions as Bt, surfaceHash as C, loadEvalFixture as Cn, tangleTracesRoot as Ct, FileSearchLedger as D, AnalyzeCrossSurfaceInteractionsInput as Dn, ScoreboardRenderOptions as Dt, acquireSingleRunLock as E, analyzeCrossSurfaceInteractions as En, PlaybackStep as Et, SearchCandidateRegisteredEvent as F, CrossSurfaceCandidateEvidence as Fn, scoreboardSummary as Ft, SearchLedgerEvent as G, CrossSurfaceInteractionAwareSelection as Gn, BackendIntegrityReport as Gt, SearchLedger as H, CrossSurfaceEligibility as Hn, ScenarioRollup as Ht, SearchCandidateSlot as I, CrossSurfaceCandidateOutcome as In, userStoryScoreboard as It, SearchLedgerTrustedHeadMode as J, CrossSurfaceInteractionReport as Jn, summarizeAgentReceiptIntegrity as Jt, SearchLedgerHash as K, CrossSurfaceInteractionEffect as Kn, assertRealAgentReceipts as Kt, SearchCandidateSlotClosedEvent as L, CrossSurfaceCandidateSummary as Ln, ProfileDispatchFn as Lt, SearchAttemptAccounting as M, CrossSurfaceBootstrapPolicy as Mn, makePlaybackDispatch as Mt, SearchCandidateDecidedEvent as N, CrossSurfaceCandidate as Nn, renderScoreboardMarkdown as Nt, OpenSearchLedgerOptions as O, CrossSurfaceAdditionDecision as On, ScoreboardRow as Ot, SearchCandidateLineage as P, CrossSurfaceCandidateComparison as Pn, scoreUserStory as Pt, SearchPlan as Q, CrossSurfacePairEvidence as Qn, RuntimeEventLike as Qt, SearchCandidateSurface as R, CrossSurfaceComponent as Rn, ProfileMatrixError as Rt, surfaceContentHash as S, discoverEvalFixtures as Sn, resolveRunDir as St, SingleRunLockOptions as T, planEvalFixtureRun as Tn, PlaybackDriver as Tt, SearchLedgerAppendResult as U, CrossSurfaceEvidenceBreakdown as Un, runProfileMatrix as Ut, SearchFailureReason as V, CrossSurfaceDistribution as Vn, RunProfileMatrixResult as Vt, SearchLedgerEntry as W, CrossSurfaceIneligibilityReason as Wn, BackendIntegrityError as Wt, SearchOperationKind as X, CrossSurfaceNaiveStackSelection as Xn, ArtifactEventLike as Xt, SearchModelIdentity as Y, CrossSurfaceInteractionTask as Yn, summarizeBackendIntegrity as Yt, SearchOperationRecordedEvent as Z, CrossSurfacePairCompatibility as Zn, ProposalEventLike as Zt, assertCodeSurfaceIdentity as _, EvalFixtureRunPlan as _n, compareRankKeys as _t, WorktreeAdapterError as a, RolloutArgumentDiff as an, CrossSurfaceTaskRow as ar, SearchSurfaceKind as at, componentSurfaceIdentityMaterial as b, LoadEvalFixtureScenariosOptions as bn, scoreDiscrimination as bt, verifyCodeSurface as c, ScoredRollout as cn, TraceAnalystScenario as cr, SearchTokenAccounting as ct, PhoenixEvaluationResultLike as d, rolloutArgumentDiff as dn, SearchLedgerConflictError as dt, extractProducedState as en, CrossSurfacePairwiseEntry as er, SearchPlannedOperation as et, PhoenixEvaluatorLike as f, NeutralizationGateOptions as fn, SearchLedgerError as ft, isTransientTransportFailure as g, EvalFixtureLoadOptions as gn, campaignMeanComposite as gt, TransientFailureOptions as h, EvalFixtureFile as hn, campaignBreakdown as ht, WorktreeAdapter as i, LabeledScenarioStoreError as in, CrossSurfaceSelections as ir, SearchSurfaceEvidence as it, SearchArtifactRef as j, CrossSurfaceBestSingleSelection as jn, UserStoryVerdict as jt, SEARCH_LEDGER_SCHEMA as k, CrossSurfaceAdditionRejectionReason as kn, ScoreboardSummary as kt, AutoevalsScoreLike as l, UngroundedLiteralReport as ln, buildTraceAnalystSurfaceDispatch as lr, openSearchLedger as lt, phoenixEvaluatorJudge as m, EvalFixture as mn, CampaignBreakdown as mt, GitWorktreeAdapterOptions as n, FsLabeledScenarioStore as nn, CrossSurfaceRelativeCost as nr, SearchSourceRef as nt, gitWorktreeAdapter as o, RolloutArgumentDiffOptions as on, BuildTraceAnalystSurfaceDispatchOptions as or, SearchTaskAttemptedEvent as ot, autoevalsScorerJudge as p, neutralizationGate as pn, SearchLedgerIntegrityError as pt, SearchLedgerReplay as q, CrossSurfaceInteractionPath as qn, assertRealBackend as qt, Worktree as r, FsLabeledScenarioStoreOptions as rn, CrossSurfaceSelectionPolicy as rr, SearchSurfaceEffect as rt, resolveWorktreePath as s, RolloutCall as sn, TraceAnalystArtifact as sr, SearchTaskOutcome as st, CodeSurfaceVerification as t, neutralizeText as tn, CrossSurfaceRankedSingle as tr, SearchPlannedTask as tt, AutoevalsScorerLike as u, classifyUngroundedLiterals as un, traceAnalystQualityJudge as ur, validateSearchLedgerEvent as ut, assertComponentSurface as v, EvalFixtureScenario as vn, DiscriminationScore as vt, SingleRunLock as w, loadEvalFixtureScenarios as wn, PlaybackContext as wt, renderSurfaceDiff as x, PlanEvalFixtureRunOptions as xn, selectDiscriminative as xt, codeSurfaceIdentityMaterial as y, EvalFixtureValidationMode as yn, ScenarioSignal as yt, SearchCompletedEvent as z, CrossSurfaceComponentEvidence as zn, ProfileSummary as zt };
|
|
1515
|
+
//# sourceMappingURL=index-Sh2I0DRc.d.ts.map
|