@tangle-network/agent-eval 0.144.5 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +471 -89
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +8 -6
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
- package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
- package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +11 -10
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
- package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
- package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
- package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
- package/dist/errors-D-LKuDhb.js.map +1 -1
- package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
- package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
- package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
- package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
- package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +252 -451
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +271 -751
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
- package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
- package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
- package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +2 -2
- package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
- package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +19 -9
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
- package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
- package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
- package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
- package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
- package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
- package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
- package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/building-doctrine.md +15 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +21 -4
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
- package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
- package/dist/campaign-ClpnD7Ug.js.map +0 -1
- package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
- package/dist/index-CpxZSlB7.d.ts.map +0 -1
- package/dist/index-CsuAo2-J.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-D3EoChAU.js.map +0 -1
- package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
- package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
- package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
- package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
- package/dist/types-BjMFz88h.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
package/dist/index.js
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
|
|
2
|
-
import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-
|
|
2
|
+
import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-qgWLA6E9.js";
|
|
3
3
|
import { a as LimitExceededError, c as ValidationError, i as JudgeError, l as VerificationError, n as CaptureIntegrityError, o as NotFoundError, r as ConfigError, s as ReplayError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
|
|
4
|
-
import {
|
|
4
|
+
import { a as verifyManifest, i as signManifest, n as evaluateHypothesis, r as hashJson, t as canonicalize } from "./pre-registration-DakwTRXk.js";
|
|
5
|
+
import { AGENT_PROFILE_KINDS, AgentProfileCellValidationError, agentProfileCellHashMaterial, agentProfileCellKey, assertRunAgentProfileCell, buildAgentInterfaceProfileCell, buildAgentProfileCell, groupRunsByAgentProfileCell, requireAgentProfileCell, toAgentProfileJson, validateAgentProfileCell, verifyAgentProfileCell } from "./profile-cell.js";
|
|
5
6
|
import { a as estimateTokens, i as estimateCost, n as MetricsCollector, o as isModelPriced, r as TokenCounter, s as resolveModelPricing, t as MODEL_PRICING } from "./metrics-C9YY1OcL.js";
|
|
6
7
|
import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-DMFxsLKr.js";
|
|
8
|
+
import { C as servedModelAcceptable, E as judgeFamily, S as normalizeModelId, T as assertCrossFamily, _ as ServedCrossFamilyError, a as assertLlmRoute, b as assertServedModels, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, g as PROBE_MAX_TOKENS, h as ModelSubstitutionError, i as LlmRouteAssertionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError, v as assertCrossFamilyServed, w as CrossFamilyError, x as checkServedModel, y as assertServedModel } from "./llm-client-DzvMUsS_.js";
|
|
7
9
|
import { a as providerFromBaseUrl, i as defaultProviderRedactor, n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
|
|
8
|
-
import { a as
|
|
9
|
-
import { C as createChatClient, S as computeTraceMetrics, _ as CONTROL_INTEGRITY_ANALYST, a as analystFindingDigest, c as completedAnalystReviewQuality, d as validateAnalystReviewDecisions, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, l as readAnalystReview, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, o as analystRunDigest, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, s as assertUniqueFindingIds, t as buildDefaultAnalystRegistry, u as snapshotAnalystRun, v as ControlIntegrityAnalyst, y as emitControlIntegrityFindings } from "./default-registry-RLNNoeEP.js";
|
|
10
|
+
import { C as createChatClient, S as computeTraceMetrics, _ as CONTROL_INTEGRITY_ANALYST, a as analystFindingDigest, c as completedAnalystReviewQuality, d as validateAnalystReviewDecisions, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, l as readAnalystReview, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, o as analystRunDigest, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, s as assertUniqueFindingIds, t as buildDefaultAnalystRegistry, u as snapshotAnalystRun, v as ControlIntegrityAnalyst, y as emitControlIntegrityFindings } from "./default-registry-BaQXW1Ow.js";
|
|
10
11
|
import { INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OUTPUT_VALUE, RUN_COST_ATTR_KEYS, SPAN_KIND_ATTR_KEYS, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, applyLlmSpanOtlpAttributes, asNumber, contextInputTokens, firstNumberAttr } from "./trace-attributes.js";
|
|
11
12
|
import { C as TraceNotFoundError, G as resolveTraceAnalystLimits, J as extractOtlpAttributes, K as asString, Q as readOtlpStatus, S as TraceFileTooLargeError, W as DEFAULT_TRACE_ANALYST_LIMITS, X as inferOtlpKind, Y as firstStringAttr, Z as projectOtlpFlatLine, _ as TraceAnalysisLimitError, b as TraceFileMalformedError, c as traceAnalystFunctionGroup, et as stringField, g as SpanNotFoundError, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst, it as traceSpanKindToOpenInferenceKind, l as createBoundedTraceAnalysisStore, m as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, n as renderPriorFindings, nt as classifyOtlpSpanRole, o as TRACE_ANALYST_TOOL_NAMESPACE, p as DEFAULT_TRACE_ANALYST_BUDGETS, r as renderUpstreamFindings, rt as isOtlpModelCall, s as buildTraceAnalysisToolDescriptors, t as createTraceAnalyst, tt as applyToolSpanOtlpAttributes, v as TraceAnalysisStoreContractError, x as TraceFileMissingError, y as TraceAnalysisValidationError } from "./kind-factory-BHIgPmzS.js";
|
|
12
13
|
import { a as computeFindingId, o as makeFinding, s as makeProposalFinding } from "./usage-receipt-EVI8B8Xu.js";
|
|
@@ -14,42 +15,46 @@ import { c as SUPERVISOR_RUN_INTEGRITY_SCHEMA, n as supervisorRunRolloutLines, t
|
|
|
14
15
|
import { c as assertMintedLines, f as isRolloutLine, i as ROLLOUT_SCHEMA, m as validateRolloutLine, p as isTrainableSplit, s as assertMinted, u as assertRolloutLine } from "./schema-C6DW4ZHR.js";
|
|
15
16
|
import { a as scoreOrigin, n as observedScore, o as trainingReward, r as observedSplitScore, s as trainingScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
|
|
16
17
|
import { a as clamp01, i as aggregateRunScore, r as DEFAULT_RUN_SCORE_WEIGHTS } from "./proposal-findings-2GIUo1et.js";
|
|
17
|
-
import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-
|
|
18
|
-
import {
|
|
19
|
-
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-
|
|
18
|
+
import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-Bmrq6yqU.js";
|
|
19
|
+
import { a as inMemoryVerdictCache, i as fileVerdictCache, n as canonicalJson, r as contentHash, t as cachedJudge } from "./verdict-cache-BCcOh0kF.js";
|
|
20
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-BiN49gK6.js";
|
|
20
21
|
import { f as Mutex, p as mapConcurrent } from "./ledger-core-DXZIqu17.js";
|
|
21
|
-
import {
|
|
22
|
+
import { $ as runReferenceEquivalenceJudge, L as DEFAULT_MUTATION_PRIMITIVES, M as surfaceContentHash, Q as createReferenceEquivalenceJudge, R as buildReflectionPrompt, X as REFERENCE_EQUIVALENCE_INPUT_LIMITS, Z as REFERENCE_EQUIVALENCE_JUDGE_VERSION, _t as assertRealBackend, at as Dataset, bt as JudgeParseError, ct as llmJudge, dt as paretoFrontier, et as DEFAULT_RED_TEAM_CORPUS, ft as paretoFrontierWithCrowding, gt as assertRealAgentReceipts, ht as BackendIntegrityError, it as toolNamesForRun, lt as crowdingDistance, nt as redTeamReport, ot as HoldoutLockedError, pt as scalarScore, rt as scoreRedTeamOutput, st as hashScenarios, tt as redTeamDataset, ut as dominates, vt as summarizeAgentReceiptIntegrity, y as runCanaries, yt as summarizeBackendIntegrity, z as parseReflectionResponse } from "./skillopt-optimization-method-CQwZ-ZX8.js";
|
|
22
23
|
import { $ as selfPreference, A as pairedRiskDifference, B as requiredSampleSize, C as mulberry32, D as pairedCohensDz, E as pairedBootstrap, F as partialCredit, G as wilson, H as weightedComposite, I as passAtK, J as normalCdf, K as studentTCdf, L as pearsonR, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, P as pairedTTest, Q as positionalBias, R as ranks, S as mcnemarRequiredN, T as pairedBinaryScale, U as weightedMean, V as spearmanR, W as wilcoxonSignedRank, X as calibrateJudgeContinuous, Y as calibrateJudge, Z as continuousAgreement, _ as interpretCliffs, a as MANN_WHITNEY_EXACT_MAX_WORK, b as mcnemar, c as bonferroni, d as confidenceInterval, et as verbosityBias, f as corpusInterRaterAgreement, g as interRaterReliability, h as holm, i as MANN_WHITNEY_EXACT_MAX_STATES, j as pairedRiskDifferenceExact, k as pairedMde, l as cliffsDelta, m as eProcess, n as DECISION_PAIRED_DELTA_STATISTIC, o as WILCOXON_EXACT_MAX_N, p as corpusInterRaterAgreementFromJudgeScores, r as DEFAULT_PERMUTATIONS, s as benjaminiHochberg, t as BOOTSTRAP_GATE_MIN_N, u as cohensD, v as isBinaryOutcomeVector, w as normalizeScores, x as mcnemarPower, y as mannWhitneyU, z as requiredPairedSampleSize } from "./statistics-ByxzSiOM.js";
|
|
23
24
|
import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-iZ08VFMN.js";
|
|
25
|
+
import { a as improvementVerdict, i as gitProvenanceReader, n as computeExperimentStats, o as inMemoryExperimentStore, r as fileExperimentStore, s as clusteredPairedBinary, t as ExperimentTracker } from "./experiment-tracker-CnRICnMl.js";
|
|
24
26
|
import { n as llmSpanFromProvider, t as TraceEmitter } from "./emitter-CPBAhxum.js";
|
|
25
27
|
import { a as isRetrievalSpan, i as isLlmSpan, n as TRACE_SCHEMA_VERSION, o as isSandboxSpan, r as isJudgeSpan, s as isToolSpan, t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
|
|
26
|
-
import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-
|
|
27
|
-
import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-
|
|
28
|
+
import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-DqOw5X6_.js";
|
|
29
|
+
import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-Dy39cbFF.js";
|
|
30
|
+
import { d as pairedDecisionShape, f as minimumPairsForPairedDeltaTest, p as pairedDeltaTest, u as decidePairedPromotion } from "./promotion-policy-CrLrmys8.js";
|
|
28
31
|
import { i as toRewardRows, r as toJsonl, s as toSftRows } from "./exporters-q9iL-2Jf.js";
|
|
29
|
-
import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-
|
|
32
|
+
import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-2ECTXb2N.js";
|
|
30
33
|
import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
|
|
31
|
-
import { n as unmintableReasons, t as mintRolloutRows } from "./mint-
|
|
32
|
-
import { C as rollupSupervisorRuns, D as isUnavailable, E as SUPERVISOR_RUN_SCHEMA, O as showMeasured, b as readClaudeCodeSupervisorRun, c as writeSupervisorRunReport, d as readRuntimeSupervisorRun, f as runtimeSupervisorRunReader, h as renderSupervisorRunMarkdown, m as renderSupervisorRunHeadline, t as analyzeSupervisorRun, u as isRuntimeSupervisorRunDir, x as analyzeSupervisorRunSources, y as claudeCodeSupervisorRunReader } from "./supervisor-run-
|
|
33
|
-
import { A as TRACE_ANALYST_ACTOR_DESCRIPTION, C as domainEvidencePattern, D as tokenizeDomainWords, E as scoreTraceInsightReadiness, O as traceAnalystOnRunComplete, S as describeTraceInsightScope, T as planTraceInsightQuestions, _ as otlpToTraceRunRecords, a as convertTraceStoresToOtlp, b as buildTraceInsightPrompt, c as otelRunCompleteHook, d as captureFetchToRawSink, f as ToolTraceMissingError, g as otlpToRunRecords, h as otlpRowsToTraceRunRecords, i as iterateRawCalls, j as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, k as analyzeTraces, l as OTEL_AGENT_EVAL_SCOPE, m as otlpRowsToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as toolSpansToTraceAnalysisStore, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as flattenOtlpExportToNdjson, w as inferDomainKeywords, x as defaultTraceInsightPanel, y as buildTraceInsightContext } from "./replay-
|
|
34
|
+
import { n as unmintableReasons, t as mintRolloutRows } from "./mint-CGEkzPLf.js";
|
|
35
|
+
import { C as rollupSupervisorRuns, D as isUnavailable, E as SUPERVISOR_RUN_SCHEMA, O as showMeasured, b as readClaudeCodeSupervisorRun, c as writeSupervisorRunReport, d as readRuntimeSupervisorRun, f as runtimeSupervisorRunReader, h as renderSupervisorRunMarkdown, m as renderSupervisorRunHeadline, t as analyzeSupervisorRun, u as isRuntimeSupervisorRunDir, x as analyzeSupervisorRunSources, y as claudeCodeSupervisorRunReader } from "./supervisor-run-D_sokXcO.js";
|
|
36
|
+
import { A as TRACE_ANALYST_ACTOR_DESCRIPTION, C as domainEvidencePattern, D as tokenizeDomainWords, E as scoreTraceInsightReadiness, O as traceAnalystOnRunComplete, S as describeTraceInsightScope, T as planTraceInsightQuestions, _ as otlpToTraceRunRecords, a as convertTraceStoresToOtlp, b as buildTraceInsightPrompt, c as otelRunCompleteHook, d as captureFetchToRawSink, f as ToolTraceMissingError, g as otlpToRunRecords, h as otlpRowsToTraceRunRecords, i as iterateRawCalls, j as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, k as analyzeTraces, l as OTEL_AGENT_EVAL_SCOPE, m as otlpRowsToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as toolSpansToTraceAnalysisStore, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as flattenOtlpExportToNdjson, w as inferDomainKeywords, x as defaultTraceInsightPanel, y as buildTraceInsightContext } from "./replay-DyBLaKFc.js";
|
|
34
37
|
import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-CKtTpRhv.js";
|
|
35
38
|
import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BW27f3XW.js";
|
|
36
|
-
import { B as
|
|
39
|
+
import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-Tdy3h62h.js";
|
|
37
40
|
import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-C-GocmIW.js";
|
|
38
41
|
import { a as judgeSpans, c as runsForScenario, i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy, s as runFailureClass, t as aggregateLlm } from "./query-Di7eEQ79.js";
|
|
39
|
-
import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-
|
|
40
|
-
import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-
|
|
42
|
+
import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-BNNK7irB.js";
|
|
43
|
+
import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-Lf-5I7xh.js";
|
|
41
44
|
import { a as composeParsers, c as vitestTestParser, i as SubprocessSandboxDriver, n as DockerSandboxDriver, o as jestTestParser, r as SandboxHarness, s as pytestTestParser, t as runTestGradedScenario } from "./test-graded-scenario-JHcKQNpq.js";
|
|
42
|
-
import {
|
|
45
|
+
import { n as InMemoryTraceStore, t as FileSystemTraceStore } from "./store-DNe_Uv1Q.js";
|
|
46
|
+
import { n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-MLzHOfV9.js";
|
|
43
47
|
import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
44
48
|
import { n as DEFAULT_RULES, r as classifyFailure, t as computeToolUseMetrics } from "./tool-use-metrics-DEGMKycK.js";
|
|
45
49
|
import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
|
|
46
|
-
import {
|
|
47
|
-
import { t as
|
|
50
|
+
import { n as runCounterfactual, t as attributeCounterfactuals } from "./counterfactual-CWPTrMH7.js";
|
|
51
|
+
import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-CDSolHq7.js";
|
|
52
|
+
import { t as runEvalCampaign } from "./eval-campaign-DNjCvAm-.js";
|
|
48
53
|
import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
|
|
49
54
|
import { accessSync, appendFileSync, constants, cpSync, existsSync, mkdirSync, promises, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
|
|
50
55
|
import { basename, delimiter, dirname, extname, isAbsolute, join, relative, resolve } from "node:path";
|
|
51
56
|
import { createHash } from "node:crypto";
|
|
52
|
-
import {
|
|
57
|
+
import { spawnSync } from "node:child_process";
|
|
53
58
|
import { readFile } from "node:fs/promises";
|
|
54
59
|
import { cpus } from "node:os";
|
|
55
60
|
import { gzipSync } from "node:zlib";
|
|
@@ -423,6 +428,10 @@ async function executeScenario(chat, scenario, config) {
|
|
|
423
428
|
receiptFromError: costReceiptFromLlmError
|
|
424
429
|
});
|
|
425
430
|
if (!paid.succeeded) throw paid.error;
|
|
431
|
+
assertServedModel(model, paid.value.servedModel, {
|
|
432
|
+
allowUnreported: true,
|
|
433
|
+
context: `executeScenario "${scenario.id}" turn ${i}`
|
|
434
|
+
});
|
|
426
435
|
const rawContent = paid.value.content;
|
|
427
436
|
if (typeof rawContent !== "string") throw new CaptureIntegrityError(`chat response for scenario "${scenario.id}" turn ${i} is malformed: expected content to be a string, got ${rawContent === null ? "null" : typeof rawContent}`);
|
|
428
437
|
const content = rawContent;
|
|
@@ -1100,235 +1109,6 @@ async function runE2EWorkflow(client, name, workflow) {
|
|
|
1100
1109
|
};
|
|
1101
1110
|
}
|
|
1102
1111
|
//#endregion
|
|
1103
|
-
//#region src/clustered-paired-binary.ts
|
|
1104
|
-
/**
|
|
1105
|
-
* Paired binary comparison for work items nested inside independent clusters.
|
|
1106
|
-
*
|
|
1107
|
-
* Pairing is delegated to {@link pairArms}; this module adds the cluster-aware
|
|
1108
|
-
* estimands and inference that task-level McNemar/bootstrap utilities cannot
|
|
1109
|
-
* provide. Callers keep their own row shape through accessors, and every
|
|
1110
|
-
* matched or unpaired result returns the original row object unchanged.
|
|
1111
|
-
*/
|
|
1112
|
-
const DEFAULT_BOOTSTRAP_RESAMPLES = 1e4;
|
|
1113
|
-
const DEFAULT_SIGN_FLIP_RESAMPLES = 1e5;
|
|
1114
|
-
const MAX_RESAMPLES = 1e6;
|
|
1115
|
-
const DEFAULT_EXACT_CLUSTER_LIMIT = 20;
|
|
1116
|
-
const SIGN_FLIP_SEED_SALT = 2654435769;
|
|
1117
|
-
/**
|
|
1118
|
-
* Compare binary outcomes on matched work items while respecting independent
|
|
1119
|
-
* clusters. The confidence interval resamples whole clusters and recomputes the
|
|
1120
|
-
* task-weighted risk difference. The sign-flip test flips whole-cluster outcome
|
|
1121
|
-
* totals and tests that same task-weighted estimand.
|
|
1122
|
-
*/
|
|
1123
|
-
function clusteredPairedBinary(rows, options) {
|
|
1124
|
-
const config = validateOptions(options);
|
|
1125
|
-
const paired = pairArms(projectSelectedRows(rows, options), {
|
|
1126
|
-
baselineArm: options.baselineArm,
|
|
1127
|
-
treatmentArm: options.treatmentArm
|
|
1128
|
-
});
|
|
1129
|
-
const matchedPairs = paired.pairs.map((pair) => {
|
|
1130
|
-
const baseline = pair.baseline;
|
|
1131
|
-
const treatment = pair.treatment;
|
|
1132
|
-
if (baseline.clusterKey !== treatment.clusterKey) throw new ValidationError(`clusteredPairedBinary: pairKey '${pair.pairKey}' rep ${pair.repIndex} crosses clusters ('${baseline.clusterKey}' vs '${treatment.clusterKey}')`);
|
|
1133
|
-
return {
|
|
1134
|
-
pairKey: pair.pairKey,
|
|
1135
|
-
repIndex: pair.repIndex,
|
|
1136
|
-
clusterKey: baseline.clusterKey,
|
|
1137
|
-
baseline: baseline.original,
|
|
1138
|
-
treatment: treatment.original,
|
|
1139
|
-
baselinePass: baseline.pass,
|
|
1140
|
-
treatmentPass: treatment.pass
|
|
1141
|
-
};
|
|
1142
|
-
});
|
|
1143
|
-
const unpairedBaseline = paired.unpairedBaseline.map((row) => row.original);
|
|
1144
|
-
const unpairedTreatment = paired.unpairedTreatment.map((row) => row.original);
|
|
1145
|
-
if (matchedPairs.length === 0) return {
|
|
1146
|
-
matchedPairs,
|
|
1147
|
-
unpairedBaseline,
|
|
1148
|
-
unpairedTreatment,
|
|
1149
|
-
statistics: null
|
|
1150
|
-
};
|
|
1151
|
-
const clusters = summarizeClusters(matchedPairs);
|
|
1152
|
-
const b10 = clusters.reduce((sum, cluster) => sum + cluster.b10, 0);
|
|
1153
|
-
const b01 = clusters.reduce((sum, cluster) => sum + cluster.b01, 0);
|
|
1154
|
-
const taskWeightedRiskDifference = (b10 - b01) / matchedPairs.length;
|
|
1155
|
-
const equalClusterMean = mean$4(clusters.map((cluster) => cluster.meanDifference));
|
|
1156
|
-
const bootstrap = clusters.length < 2 ? null : clusterBootstrap(clusters, config);
|
|
1157
|
-
const signFlip = clusterSignFlip(clusters, config);
|
|
1158
|
-
return {
|
|
1159
|
-
matchedPairs,
|
|
1160
|
-
unpairedBaseline,
|
|
1161
|
-
unpairedTreatment,
|
|
1162
|
-
statistics: {
|
|
1163
|
-
nPairs: matchedPairs.length,
|
|
1164
|
-
nClusters: clusters.length,
|
|
1165
|
-
b10,
|
|
1166
|
-
b01,
|
|
1167
|
-
taskWeightedRiskDifference,
|
|
1168
|
-
equalClusterMean,
|
|
1169
|
-
clusters,
|
|
1170
|
-
bootstrap,
|
|
1171
|
-
signFlip
|
|
1172
|
-
}
|
|
1173
|
-
};
|
|
1174
|
-
}
|
|
1175
|
-
function validateOptions(options) {
|
|
1176
|
-
assertNonEmptyString("baselineArm", options.baselineArm);
|
|
1177
|
-
assertNonEmptyString("treatmentArm", options.treatmentArm);
|
|
1178
|
-
if (options.baselineArm === options.treatmentArm) throw new ValidationError(`clusteredPairedBinary: baselineArm and treatmentArm are both '${options.baselineArm}'`);
|
|
1179
|
-
if (!Number.isInteger(options.seed)) throw new ValidationError(`clusteredPairedBinary: seed must be an integer, got ${options.seed}`);
|
|
1180
|
-
const confidence = options.confidence ?? .95;
|
|
1181
|
-
if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new ValidationError(`clusteredPairedBinary: confidence must be in (0,1), got ${confidence}`);
|
|
1182
|
-
const bootstrapResamples = options.bootstrapResamples ?? DEFAULT_BOOTSTRAP_RESAMPLES;
|
|
1183
|
-
assertResampleCount("bootstrapResamples", bootstrapResamples);
|
|
1184
|
-
const rawMinimumBootstrapResamples = 2 / (1 - confidence);
|
|
1185
|
-
const minimumBootstrapResamples = Math.ceil(rawMinimumBootstrapResamples - Number.EPSILON * Math.max(1, rawMinimumBootstrapResamples) * 8);
|
|
1186
|
-
if (bootstrapResamples < minimumBootstrapResamples) throw new ValidationError(`clusteredPairedBinary: bootstrapResamples must be at least ${minimumBootstrapResamples} for confidence ${confidence} so both interval tails are represented, got ${bootstrapResamples}`);
|
|
1187
|
-
const signFlipResamples = options.signFlipResamples ?? DEFAULT_SIGN_FLIP_RESAMPLES;
|
|
1188
|
-
assertResampleCount("signFlipResamples", signFlipResamples);
|
|
1189
|
-
const exactClusterLimit = options.exactClusterLimit ?? DEFAULT_EXACT_CLUSTER_LIMIT;
|
|
1190
|
-
if (!Number.isInteger(exactClusterLimit) || exactClusterLimit < 0 || exactClusterLimit > DEFAULT_EXACT_CLUSTER_LIMIT) throw new ValidationError(`clusteredPairedBinary: exactClusterLimit must be an integer in [0,${DEFAULT_EXACT_CLUSTER_LIMIT}], got ${exactClusterLimit}`);
|
|
1191
|
-
const alternative = options.alternative ?? "two-sided";
|
|
1192
|
-
if (alternative !== "two-sided" && alternative !== "greater" && alternative !== "less") throw new ValidationError(`clusteredPairedBinary: alternative must be 'two-sided', 'greater', or 'less', got ${String(alternative)}`);
|
|
1193
|
-
return {
|
|
1194
|
-
seed: options.seed,
|
|
1195
|
-
confidence,
|
|
1196
|
-
bootstrapResamples,
|
|
1197
|
-
alternative,
|
|
1198
|
-
exactClusterLimit,
|
|
1199
|
-
signFlipResamples
|
|
1200
|
-
};
|
|
1201
|
-
}
|
|
1202
|
-
function assertResampleCount(name, value) {
|
|
1203
|
-
if (!Number.isInteger(value) || value <= 0) throw new ValidationError(`clusteredPairedBinary: ${name} must be a positive integer, got ${value}`);
|
|
1204
|
-
if (value > MAX_RESAMPLES) throw new ValidationError(`clusteredPairedBinary: ${name} must not exceed ${MAX_RESAMPLES}, got ${value}`);
|
|
1205
|
-
}
|
|
1206
|
-
function projectSelectedRows(rows, options) {
|
|
1207
|
-
const projected = [];
|
|
1208
|
-
for (const original of rows) {
|
|
1209
|
-
const arm = options.arm(original);
|
|
1210
|
-
assertNonEmptyString("arm", arm);
|
|
1211
|
-
if (arm !== options.baselineArm && arm !== options.treatmentArm) continue;
|
|
1212
|
-
const pairKey = options.pairKey(original);
|
|
1213
|
-
const clusterKey = options.clusterKey(original);
|
|
1214
|
-
const pass = options.pass(original);
|
|
1215
|
-
const repKey = options.repKey?.(original);
|
|
1216
|
-
assertNonEmptyString("pairKey", pairKey);
|
|
1217
|
-
assertNonEmptyString("clusterKey", clusterKey);
|
|
1218
|
-
if (typeof pass !== "boolean") throw new ValidationError(`clusteredPairedBinary: pass accessor must return boolean for pairKey '${pairKey}'`);
|
|
1219
|
-
if (repKey !== void 0) assertNonEmptyString("repKey", repKey);
|
|
1220
|
-
projected.push({
|
|
1221
|
-
pairKey,
|
|
1222
|
-
clusterKey,
|
|
1223
|
-
arm,
|
|
1224
|
-
pass,
|
|
1225
|
-
repKey,
|
|
1226
|
-
original
|
|
1227
|
-
});
|
|
1228
|
-
}
|
|
1229
|
-
return projected;
|
|
1230
|
-
}
|
|
1231
|
-
function assertNonEmptyString(name, value) {
|
|
1232
|
-
if (typeof value !== "string" || value.trim().length === 0) throw new ValidationError(`clusteredPairedBinary: ${name} accessor must return a non-empty string`);
|
|
1233
|
-
}
|
|
1234
|
-
function summarizeClusters(pairs) {
|
|
1235
|
-
const byCluster = /* @__PURE__ */ new Map();
|
|
1236
|
-
for (const pair of pairs) {
|
|
1237
|
-
const summary = byCluster.get(pair.clusterKey) ?? {
|
|
1238
|
-
nPairs: 0,
|
|
1239
|
-
b10: 0,
|
|
1240
|
-
b01: 0
|
|
1241
|
-
};
|
|
1242
|
-
summary.nPairs++;
|
|
1243
|
-
if (pair.treatmentPass && !pair.baselinePass) summary.b10++;
|
|
1244
|
-
else if (pair.baselinePass && !pair.treatmentPass) summary.b01++;
|
|
1245
|
-
byCluster.set(pair.clusterKey, summary);
|
|
1246
|
-
}
|
|
1247
|
-
return [...byCluster.entries()].sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0).map(([clusterKey, summary]) => ({
|
|
1248
|
-
clusterKey,
|
|
1249
|
-
...summary,
|
|
1250
|
-
meanDifference: (summary.b10 - summary.b01) / summary.nPairs
|
|
1251
|
-
}));
|
|
1252
|
-
}
|
|
1253
|
-
function clusterBootstrap(clusters, config) {
|
|
1254
|
-
const rng = mulberry32(config.seed);
|
|
1255
|
-
const samples = new Array(config.bootstrapResamples);
|
|
1256
|
-
for (let draw = 0; draw < config.bootstrapResamples; draw++) {
|
|
1257
|
-
let differenceSum = 0;
|
|
1258
|
-
let pairCount = 0;
|
|
1259
|
-
for (let index = 0; index < clusters.length; index++) {
|
|
1260
|
-
const cluster = clusters[Math.floor(rng() * clusters.length)];
|
|
1261
|
-
differenceSum += cluster.b10 - cluster.b01;
|
|
1262
|
-
pairCount += cluster.nPairs;
|
|
1263
|
-
}
|
|
1264
|
-
samples[draw] = differenceSum / pairCount;
|
|
1265
|
-
}
|
|
1266
|
-
samples.sort((a, b) => a - b);
|
|
1267
|
-
const alpha = 1 - config.confidence;
|
|
1268
|
-
const lowerIndex = Math.floor(alpha / 2 * config.bootstrapResamples);
|
|
1269
|
-
const upperIndex = Math.min(config.bootstrapResamples - 1, Math.ceil((1 - alpha / 2) * config.bootstrapResamples) - 1);
|
|
1270
|
-
return {
|
|
1271
|
-
statistic: "task-weighted-risk-difference",
|
|
1272
|
-
lower: samples[lowerIndex],
|
|
1273
|
-
upper: samples[Math.max(lowerIndex, upperIndex)],
|
|
1274
|
-
confidence: config.confidence,
|
|
1275
|
-
resamples: config.bootstrapResamples,
|
|
1276
|
-
seed: config.seed
|
|
1277
|
-
};
|
|
1278
|
-
}
|
|
1279
|
-
function clusterSignFlip(clusters, config) {
|
|
1280
|
-
const clusterTotals = clusters.map((cluster) => cluster.b10 - cluster.b01);
|
|
1281
|
-
const nonZero = clusterTotals.filter((delta) => delta !== 0);
|
|
1282
|
-
const totalPairs = clusters.reduce((sum, cluster) => sum + cluster.nPairs, 0);
|
|
1283
|
-
const statistic = clusterTotals.reduce((sum, delta) => sum + delta, 0) / totalPairs;
|
|
1284
|
-
if (nonZero.length <= config.exactClusterLimit) {
|
|
1285
|
-
const assignments = 2 ** nonZero.length;
|
|
1286
|
-
let extreme = 0;
|
|
1287
|
-
for (let mask = 0; mask < assignments; mask++) {
|
|
1288
|
-
let sum = 0;
|
|
1289
|
-
for (let index = 0; index < nonZero.length; index++) sum += (mask & 2 ** index ? 1 : -1) * nonZero[index];
|
|
1290
|
-
if (isExtreme(sum / totalPairs, statistic, config.alternative)) extreme++;
|
|
1291
|
-
}
|
|
1292
|
-
return {
|
|
1293
|
-
statistic,
|
|
1294
|
-
pValue: extreme / assignments,
|
|
1295
|
-
alternative: config.alternative,
|
|
1296
|
-
method: "exact",
|
|
1297
|
-
assignments,
|
|
1298
|
-
nClusters: clusters.length,
|
|
1299
|
-
nNonZeroClusters: nonZero.length,
|
|
1300
|
-
seed: null
|
|
1301
|
-
};
|
|
1302
|
-
}
|
|
1303
|
-
const signFlipSeed = (config.seed ^ SIGN_FLIP_SEED_SALT) >>> 0;
|
|
1304
|
-
const rng = mulberry32(signFlipSeed);
|
|
1305
|
-
let extreme = 0;
|
|
1306
|
-
for (let draw = 0; draw < config.signFlipResamples; draw++) {
|
|
1307
|
-
let sum = 0;
|
|
1308
|
-
for (const delta of nonZero) sum += (rng() < .5 ? -1 : 1) * delta;
|
|
1309
|
-
if (isExtreme(sum / totalPairs, statistic, config.alternative)) extreme++;
|
|
1310
|
-
}
|
|
1311
|
-
return {
|
|
1312
|
-
statistic,
|
|
1313
|
-
pValue: (extreme + 1) / (config.signFlipResamples + 1),
|
|
1314
|
-
alternative: config.alternative,
|
|
1315
|
-
method: "monte-carlo",
|
|
1316
|
-
assignments: config.signFlipResamples,
|
|
1317
|
-
nClusters: clusters.length,
|
|
1318
|
-
nNonZeroClusters: nonZero.length,
|
|
1319
|
-
seed: signFlipSeed
|
|
1320
|
-
};
|
|
1321
|
-
}
|
|
1322
|
-
function isExtreme(candidate, observed, alternative) {
|
|
1323
|
-
const tolerance = Number.EPSILON * Math.max(1, Math.abs(observed)) * 16;
|
|
1324
|
-
if (alternative === "greater") return candidate >= observed - tolerance;
|
|
1325
|
-
if (alternative === "less") return candidate <= observed + tolerance;
|
|
1326
|
-
return Math.abs(candidate) >= Math.abs(observed) - tolerance;
|
|
1327
|
-
}
|
|
1328
|
-
function mean$4(values) {
|
|
1329
|
-
return values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
1330
|
-
}
|
|
1331
|
-
//#endregion
|
|
1332
1112
|
//#region src/convergence.ts
|
|
1333
1113
|
/**
|
|
1334
1114
|
* ConvergenceTracker — tracks completion percentage over turns.
|
|
@@ -1606,6 +1386,10 @@ async function decideNextUserTurn(chat, opts) {
|
|
|
1606
1386
|
receiptFromError: costReceiptFromLlmError
|
|
1607
1387
|
});
|
|
1608
1388
|
if (!paid.succeeded) throw paid.error;
|
|
1389
|
+
assertServedModel(model, paid.value.servedModel, {
|
|
1390
|
+
allowUnreported: true,
|
|
1391
|
+
context: "decideNextUserTurn"
|
|
1392
|
+
});
|
|
1609
1393
|
return paid.value.content.trim();
|
|
1610
1394
|
}
|
|
1611
1395
|
//#endregion
|
|
@@ -2132,14 +1916,28 @@ function canonicalize$1(value) {
|
|
|
2132
1916
|
* - membership (free): GET `{baseUrl}/models` once; a model is `listed` when
|
|
2133
1917
|
* its id is in the served set.
|
|
2134
1918
|
* - probe (spends a tiny number of tokens): POST `{baseUrl}/chat/completions`
|
|
2135
|
-
* per model with a 1-message,
|
|
2136
|
-
* router
|
|
2137
|
-
* captured in `detail
|
|
1919
|
+
* per model with a 1-message, `PROBE_MAX_TOKENS`-token request; `served` is
|
|
1920
|
+
* whether the router reached a provider, with the HTTP `status` and the
|
|
1921
|
+
* body's `error.message` captured in `detail`, and `servedModel` recording
|
|
1922
|
+
* WHICH model answered.
|
|
1923
|
+
*
|
|
1924
|
+
* A 2xx is not proof the requested model answered — a gateway can accept one
|
|
1925
|
+
* id and route to another. The probe therefore compares the echoed id against
|
|
1926
|
+
* the requested one and reports `substitution`; `assertModelsServed` fails on
|
|
1927
|
+
* a substituted id exactly as it fails on a dead one, because a campaign that
|
|
1928
|
+
* runs on a silently-swapped model produces per-model numbers about a model it
|
|
1929
|
+
* never called.
|
|
2138
1930
|
*
|
|
2139
1931
|
* A default model the router cannot serve is a config bug. Gate a campaign on
|
|
2140
|
-
* `assertModelsServed` and it surfaces every dead id with its
|
|
2141
|
-
* instead of silently producing a stub run.
|
|
1932
|
+
* `assertModelsServed` and it surfaces every dead or substituted id with its
|
|
1933
|
+
* status + detail instead of silently producing a stub or mislabelled run.
|
|
1934
|
+
*/
|
|
1935
|
+
/**
|
|
1936
|
+
* Provider signature for "the budget ran out before an answer". The model is
|
|
1937
|
+
* alive — a provider took the request and consumed the budget — so this must
|
|
1938
|
+
* never be scored as a dead id.
|
|
2142
1939
|
*/
|
|
1940
|
+
const REASONING_BUDGET_EXHAUSTED = /reasoning[\s_-]?budget[\s_-]?exhausted/i;
|
|
2143
1941
|
function stripSlash$1(url) {
|
|
2144
1942
|
return url.replace(/\/+$/, "");
|
|
2145
1943
|
}
|
|
@@ -2158,13 +1956,19 @@ function errorMessage(body) {
|
|
|
2158
1956
|
* fallbacks.
|
|
2159
1957
|
*
|
|
2160
1958
|
* The membership check (one GET) always runs. When `probe` is true, each model
|
|
2161
|
-
* additionally gets a
|
|
1959
|
+
* additionally gets a small chat probe so a model that is listed but
|
|
2162
1960
|
* unconfigured (a 401 `model_not_found` from the router) is caught.
|
|
2163
1961
|
*/
|
|
2164
1962
|
async function preflightModels(opts) {
|
|
2165
1963
|
const fetchImpl = opts.fetchImpl ?? fetch;
|
|
2166
1964
|
const baseUrl = stripSlash$1(opts.baseUrl);
|
|
2167
1965
|
const authHeaders = { authorization: `Bearer ${opts.apiKey}` };
|
|
1966
|
+
const maxTokens = opts.probeMaxTokens ?? 64;
|
|
1967
|
+
if (!Number.isInteger(maxTokens) || maxTokens <= 0) return {
|
|
1968
|
+
succeeded: false,
|
|
1969
|
+
value: null,
|
|
1970
|
+
error: `preflightModels: probeMaxTokens must be a positive integer, got ${maxTokens}`
|
|
1971
|
+
};
|
|
2168
1972
|
let served;
|
|
2169
1973
|
try {
|
|
2170
1974
|
const res = await fetchImpl(`${baseUrl}/models`, {
|
|
@@ -2198,7 +2002,9 @@ async function preflightModels(opts) {
|
|
|
2198
2002
|
listed,
|
|
2199
2003
|
served: null,
|
|
2200
2004
|
status: null,
|
|
2201
|
-
detail: null
|
|
2005
|
+
detail: null,
|
|
2006
|
+
budgetExhausted: false,
|
|
2007
|
+
substitution: null
|
|
2202
2008
|
});
|
|
2203
2009
|
continue;
|
|
2204
2010
|
}
|
|
@@ -2215,17 +2021,29 @@ async function preflightModels(opts) {
|
|
|
2215
2021
|
role: "user",
|
|
2216
2022
|
content: "ping"
|
|
2217
2023
|
}],
|
|
2218
|
-
max_tokens:
|
|
2024
|
+
max_tokens: maxTokens
|
|
2219
2025
|
})
|
|
2220
2026
|
});
|
|
2221
2027
|
let detail = null;
|
|
2222
|
-
|
|
2028
|
+
let substitution = null;
|
|
2029
|
+
let budgetExhausted = false;
|
|
2030
|
+
const body = await res.json().catch(() => null);
|
|
2031
|
+
if (res.ok) {
|
|
2032
|
+
const echoed = body?.model;
|
|
2033
|
+
substitution = checkServedModel(model, typeof echoed === "string" && echoed.trim() !== "" ? echoed : null);
|
|
2034
|
+
} else {
|
|
2035
|
+
detail = errorMessage(body);
|
|
2036
|
+
budgetExhausted = detail !== null && REASONING_BUDGET_EXHAUSTED.test(detail);
|
|
2037
|
+
if (budgetExhausted) substitution = checkServedModel(model, null);
|
|
2038
|
+
}
|
|
2223
2039
|
results.push({
|
|
2224
2040
|
model,
|
|
2225
2041
|
listed,
|
|
2226
|
-
served: res.ok,
|
|
2042
|
+
served: res.ok || budgetExhausted,
|
|
2227
2043
|
status: res.status,
|
|
2228
|
-
detail
|
|
2044
|
+
detail,
|
|
2045
|
+
budgetExhausted,
|
|
2046
|
+
substitution
|
|
2229
2047
|
});
|
|
2230
2048
|
} catch (err) {
|
|
2231
2049
|
return {
|
|
@@ -2254,20 +2072,28 @@ function describeFailure(r) {
|
|
|
2254
2072
|
const probeNote = r.served === false ? ` (probe ${r.status}${r.detail ? `: ${r.detail}` : ""})` : "";
|
|
2255
2073
|
return `${r.model}: not in /models${probeNote}`;
|
|
2256
2074
|
}
|
|
2257
|
-
return `${r.model}: listed but probe ${r.status}${r.detail ? ` — ${r.detail}` : ""}`;
|
|
2075
|
+
if (r.served === false) return `${r.model}: listed but probe ${r.status}${r.detail ? ` — ${r.detail}` : ""}`;
|
|
2076
|
+
if (r.budgetExhausted) return `${r.model}: alive but the probe ran out of reasoning budget (status ${r.status}${r.detail ? `: ${r.detail}` : ""}) — it echoed no model id, so identity is unproven. Raise probeMaxTokens, or pass allowUnreported to accept reachability without identity.`;
|
|
2077
|
+
const s = r.substitution;
|
|
2078
|
+
if (s?.verdict === "unreported") return `${r.model}: probe answered without echoing a model id — identity unproven`;
|
|
2079
|
+
return `${r.model}: probe answered by ${s?.served} (${s?.verdict})`;
|
|
2258
2080
|
}
|
|
2259
2081
|
/**
|
|
2260
|
-
* Throw `ModelsUnreachableError` naming EVERY model that is
|
|
2261
|
-
*
|
|
2262
|
-
*
|
|
2263
|
-
*
|
|
2264
|
-
*
|
|
2082
|
+
* Throw `ModelsUnreachableError` naming EVERY model that is unusable for a
|
|
2083
|
+
* measured run — with status, detail, and served identity per model. A model
|
|
2084
|
+
* is unusable when it is unlisted, when its probe fails, or when its probe is
|
|
2085
|
+
* answered by a different model than the one requested. Callers gate a
|
|
2086
|
+
* campaign on this before spending tokens. When the network call itself fails
|
|
2087
|
+
* the underlying outcome error is rethrown — there is no partial silent pass.
|
|
2088
|
+
*
|
|
2089
|
+
* Substitution is only detectable with `probe: true`; a membership-only check
|
|
2090
|
+
* proves an id is in the catalogue, never that the catalogue entry answers.
|
|
2265
2091
|
*/
|
|
2266
2092
|
async function assertModelsServed(opts) {
|
|
2267
2093
|
const outcome = await preflightModels(opts);
|
|
2268
2094
|
if (!outcome.succeeded || outcome.value === null) throw new ConfigError(outcome.error ?? "assertModelsServed: preflight failed without an error message");
|
|
2269
|
-
const
|
|
2270
|
-
if (
|
|
2095
|
+
const bad = outcome.value.filter((r) => !r.listed || r.served === false || r.substitution !== null && !servedModelAcceptable(r.substitution, opts));
|
|
2096
|
+
if (bad.length > 0) throw new ModelsUnreachableError(`assertModelsServed: ${bad.length}/${outcome.value.length} model(s) unusable on the router — ${bad.map(describeFailure).join("; ")}`, outcome.value);
|
|
2271
2097
|
return outcome.value;
|
|
2272
2098
|
}
|
|
2273
2099
|
//#endregion
|
|
@@ -2346,95 +2172,6 @@ function assertSingleBackend(agent, judge, opts = {}) {
|
|
|
2346
2172
|
return report;
|
|
2347
2173
|
}
|
|
2348
2174
|
//#endregion
|
|
2349
|
-
//#region src/judge-families.ts
|
|
2350
|
-
/** Explicit `provider/...` prefix → family (models.dev / OpenRouter style). */
|
|
2351
|
-
const PROVIDER_PREFIX = {
|
|
2352
|
-
anthropic: "anthropic",
|
|
2353
|
-
openai: "openai",
|
|
2354
|
-
"azure-openai": "openai",
|
|
2355
|
-
google: "google",
|
|
2356
|
-
"google-vertex": "google",
|
|
2357
|
-
meta: "meta",
|
|
2358
|
-
"meta-llama": "meta",
|
|
2359
|
-
mistral: "mistral",
|
|
2360
|
-
mistralai: "mistral",
|
|
2361
|
-
deepseek: "deepseek",
|
|
2362
|
-
xai: "xai",
|
|
2363
|
-
qwen: "qwen",
|
|
2364
|
-
alibaba: "qwen",
|
|
2365
|
-
cohere: "cohere",
|
|
2366
|
-
amazon: "amazon",
|
|
2367
|
-
bedrock: "amazon",
|
|
2368
|
-
moonshot: "moonshot",
|
|
2369
|
-
moonshotai: "moonshot",
|
|
2370
|
-
kimi: "moonshot",
|
|
2371
|
-
"kimi-code": "moonshot",
|
|
2372
|
-
zhipu: "zhipu",
|
|
2373
|
-
zhipuai: "zhipu",
|
|
2374
|
-
zai: "zhipu",
|
|
2375
|
-
"z-ai": "zhipu",
|
|
2376
|
-
glm: "zhipu"
|
|
2377
|
-
};
|
|
2378
|
-
/** Fallback model-name patterns when there's no recognised provider prefix. */
|
|
2379
|
-
const NAME_PATTERNS = [
|
|
2380
|
-
[/claude/i, "anthropic"],
|
|
2381
|
-
[/\b(gpt|davinci|babbage)\b|^o[134]\b|[-/]o[134]\b|gpt-/i, "openai"],
|
|
2382
|
-
[/gemini|palm|gemma|bison/i, "google"],
|
|
2383
|
-
[/llama/i, "meta"],
|
|
2384
|
-
[/mi(s|x)tral|codestral|magistral/i, "mistral"],
|
|
2385
|
-
[/deepseek/i, "deepseek"],
|
|
2386
|
-
[/grok/i, "xai"],
|
|
2387
|
-
[/qwen/i, "qwen"],
|
|
2388
|
-
[/command-?(r|a)?/i, "cohere"],
|
|
2389
|
-
[/\b(nova|titan)\b/i, "amazon"],
|
|
2390
|
-
[/\bkimi\b|moonshot/i, "moonshot"],
|
|
2391
|
-
[/\bglm\b|zhipu|\bz-?ai\b/i, "zhipu"]
|
|
2392
|
-
];
|
|
2393
|
-
/**
|
|
2394
|
-
* Classify a model id into its provider family. Strips a `@snapshot` suffix
|
|
2395
|
-
* and prefers an explicit `provider/...` prefix; otherwise matches the model
|
|
2396
|
-
* name. Returns `unknown` when nothing matches (callers decide whether that's
|
|
2397
|
-
* acceptable — `assertCrossFamily` counts it as its own family).
|
|
2398
|
-
*/
|
|
2399
|
-
function judgeFamily(modelId) {
|
|
2400
|
-
const id = modelId.trim().split("@")[0].toLowerCase();
|
|
2401
|
-
const slash = id.indexOf("/");
|
|
2402
|
-
if (slash > 0) {
|
|
2403
|
-
const prefix = id.slice(0, slash);
|
|
2404
|
-
const mapped = PROVIDER_PREFIX[prefix];
|
|
2405
|
-
if (mapped) return mapped;
|
|
2406
|
-
}
|
|
2407
|
-
for (const [pattern, family] of NAME_PATTERNS) if (pattern.test(id)) return family;
|
|
2408
|
-
return "unknown";
|
|
2409
|
-
}
|
|
2410
|
-
var CrossFamilyError = class extends Error {
|
|
2411
|
-
families;
|
|
2412
|
-
models;
|
|
2413
|
-
constructor(message, families, models) {
|
|
2414
|
-
super(message);
|
|
2415
|
-
this.families = families;
|
|
2416
|
-
this.models = models;
|
|
2417
|
-
this.name = "CrossFamilyError";
|
|
2418
|
-
}
|
|
2419
|
-
};
|
|
2420
|
-
/**
|
|
2421
|
-
* Throw unless the judge models span at least `minFamilies` distinct provider
|
|
2422
|
-
* families. Pass the model ids backing your judge ensemble. Fail-loud by
|
|
2423
|
-
* design — a correlated single-family ensemble silently inflates agreement.
|
|
2424
|
-
*/
|
|
2425
|
-
function assertCrossFamily(models, opts = {}) {
|
|
2426
|
-
const minFamilies = opts.minFamilies ?? 2;
|
|
2427
|
-
const families = /* @__PURE__ */ new Set();
|
|
2428
|
-
for (const m of models) {
|
|
2429
|
-
const f = judgeFamily(m);
|
|
2430
|
-
if (f === "unknown" && !opts.allowUnknown) continue;
|
|
2431
|
-
families.add(f);
|
|
2432
|
-
}
|
|
2433
|
-
const list = [...families].sort();
|
|
2434
|
-
if (list.length < minFamilies) throw new CrossFamilyError(`judge ensemble spans ${list.length} provider famil${list.length === 1 ? "y" : "ies"} (${list.join(", ") || "none"}) but ${minFamilies} required — a single-family ensemble is correlated bias, not independent signal`, list, models);
|
|
2435
|
-
return list;
|
|
2436
|
-
}
|
|
2437
|
-
//#endregion
|
|
2438
2175
|
//#region src/knowledge/readiness.ts
|
|
2439
2176
|
function scoreKnowledgeReadiness(options) {
|
|
2440
2177
|
const now = options.now ?? /* @__PURE__ */ new Date();
|
|
@@ -5115,269 +4852,6 @@ var EvalTraceStore = class {
|
|
|
5115
4852
|
}
|
|
5116
4853
|
};
|
|
5117
4854
|
//#endregion
|
|
5118
|
-
//#region src/experiment-tracker.ts
|
|
5119
|
-
/**
|
|
5120
|
-
* Experiment tracker — git-provenanced experiment log with N-rep stats and a
|
|
5121
|
-
* KEEP / REGRESSION / NOISE verdict against a parent.
|
|
5122
|
-
*
|
|
5123
|
-
* Every loop the fleet runs reduces to the same question: "I ran the candidate
|
|
5124
|
-
* N times — is the median measurably better than the parent, or is the delta
|
|
5125
|
-
* inside the noise band?" The hand-rolled copies bake a fixed score scale
|
|
5126
|
-
* (percentage points), a fixed store path (`.evolve/experiments-v2.json`), and
|
|
5127
|
-
* `execSync('git …')` straight into the module. This is the canonical version:
|
|
5128
|
-
* provenance and persistence are injected, thresholds are configurable, and the
|
|
5129
|
-
* stats + verdict are pure functions you can unit-test without a git repo or a
|
|
5130
|
-
* filesystem.
|
|
5131
|
-
*
|
|
5132
|
-
* Stats per experiment: median / mean / min / max / iqr / stddev / passRate /
|
|
5133
|
-
* n, plus a `stable` flag (`iqr < iqrUnstableAbove && stddev < stddevUnstableAbove`).
|
|
5134
|
-
*
|
|
5135
|
-
* Verdict against a parent (both must have `n >= minRepsForVerdict`):
|
|
5136
|
-
* - NOISE — the candidate is too unstable to judge (`!stable`)
|
|
5137
|
-
* - KEEP — `medianDelta > keepThreshold`
|
|
5138
|
-
* - REGRESSION — `medianDelta < -regressionThreshold`
|
|
5139
|
-
* - NOISE — otherwise (delta inside the band)
|
|
5140
|
-
* With no parent (or insufficient reps) the verdict is the neutral ITERATE.
|
|
5141
|
-
*/
|
|
5142
|
-
const DEFAULTS = {
|
|
5143
|
-
keepThreshold: 5,
|
|
5144
|
-
regressionThreshold: 5,
|
|
5145
|
-
iqrUnstableAbove: 10,
|
|
5146
|
-
stddevUnstableAbove: Number.POSITIVE_INFINITY,
|
|
5147
|
-
minRepsForVerdict: 3
|
|
5148
|
-
};
|
|
5149
|
-
function resolveThresholds(t) {
|
|
5150
|
-
const r = {
|
|
5151
|
-
...DEFAULTS,
|
|
5152
|
-
...t ?? {}
|
|
5153
|
-
};
|
|
5154
|
-
if (r.keepThreshold < 0) throw new ValidationError(`experiment-tracker: keepThreshold must be >= 0, got ${r.keepThreshold}`);
|
|
5155
|
-
if (r.regressionThreshold < 0) throw new ValidationError(`experiment-tracker: regressionThreshold must be >= 0, got ${r.regressionThreshold}`);
|
|
5156
|
-
if (r.minRepsForVerdict < 1) throw new ValidationError(`experiment-tracker: minRepsForVerdict must be >= 1, got ${r.minRepsForVerdict}`);
|
|
5157
|
-
return r;
|
|
5158
|
-
}
|
|
5159
|
-
function median$1(sorted) {
|
|
5160
|
-
const n = sorted.length;
|
|
5161
|
-
if (n === 0) return 0;
|
|
5162
|
-
const mid = Math.floor(n / 2);
|
|
5163
|
-
return n % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
|
|
5164
|
-
}
|
|
5165
|
-
/** Population standard deviation (÷n). 0 for fewer than 2 values. */
|
|
5166
|
-
function stddev(values, mean) {
|
|
5167
|
-
if (values.length < 2) return 0;
|
|
5168
|
-
const variance = values.reduce((acc, v) => acc + (v - mean) ** 2, 0) / values.length;
|
|
5169
|
-
return Math.sqrt(variance);
|
|
5170
|
-
}
|
|
5171
|
-
/**
|
|
5172
|
-
* Compute the N-rep statistics for a set of reps. Pure — no I/O. The `stable`
|
|
5173
|
-
* flag is the trust gate the verdict depends on: a sample whose spread exceeds
|
|
5174
|
-
* the configured bounds can't distinguish a real delta from run-to-run noise.
|
|
5175
|
-
*/
|
|
5176
|
-
function computeExperimentStats(reps, thresholds) {
|
|
5177
|
-
const t = resolveThresholds(thresholds);
|
|
5178
|
-
const n = reps.length;
|
|
5179
|
-
if (n === 0) return {
|
|
5180
|
-
median: 0,
|
|
5181
|
-
mean: 0,
|
|
5182
|
-
min: 0,
|
|
5183
|
-
max: 0,
|
|
5184
|
-
iqr: 0,
|
|
5185
|
-
stddev: 0,
|
|
5186
|
-
passRate: null,
|
|
5187
|
-
n: 0,
|
|
5188
|
-
stable: false
|
|
5189
|
-
};
|
|
5190
|
-
const scores = reps.map((r) => {
|
|
5191
|
-
if (!Number.isFinite(r.score)) throw new ValidationError(`experiment-tracker: rep ${r.rep} has non-finite score ${r.score}`);
|
|
5192
|
-
return r.score;
|
|
5193
|
-
});
|
|
5194
|
-
const sorted = [...scores].sort((a, b) => a - b);
|
|
5195
|
-
const mean = scores.reduce((s, v) => s + v, 0) / n;
|
|
5196
|
-
const sd = stddev(scores, mean);
|
|
5197
|
-
const spread = iqr(scores);
|
|
5198
|
-
const rated = reps.filter((r) => typeof r.passed === "boolean");
|
|
5199
|
-
const passRate = rated.length === 0 ? null : rated.filter((r) => r.passed).length / rated.length;
|
|
5200
|
-
const stable = spread < t.iqrUnstableAbove && sd < t.stddevUnstableAbove;
|
|
5201
|
-
return {
|
|
5202
|
-
median: median$1(sorted),
|
|
5203
|
-
mean,
|
|
5204
|
-
min: sorted[0],
|
|
5205
|
-
max: sorted[n - 1],
|
|
5206
|
-
iqr: spread,
|
|
5207
|
-
stddev: sd,
|
|
5208
|
-
passRate,
|
|
5209
|
-
n,
|
|
5210
|
-
stable
|
|
5211
|
-
};
|
|
5212
|
-
}
|
|
5213
|
-
/**
|
|
5214
|
-
* Verdict for a candidate against its parent. Pure — operates on already-computed
|
|
5215
|
-
* stats. KEEP/REGRESSION require both sides to have `>= minRepsForVerdict` reps
|
|
5216
|
-
* AND the candidate to be `stable`; otherwise the result is NOISE (unstable) or
|
|
5217
|
-
* ITERATE (not enough reps / no parent).
|
|
5218
|
-
*/
|
|
5219
|
-
function improvementVerdict(candidate, parent, thresholds) {
|
|
5220
|
-
const t = resolveThresholds(thresholds);
|
|
5221
|
-
if (!parent) return {
|
|
5222
|
-
verdict: "ITERATE",
|
|
5223
|
-
medianDelta: null,
|
|
5224
|
-
reason: "no parent experiment to compare against"
|
|
5225
|
-
};
|
|
5226
|
-
if (candidate.n < t.minRepsForVerdict || parent.n < t.minRepsForVerdict) return {
|
|
5227
|
-
verdict: "ITERATE",
|
|
5228
|
-
medianDelta: null,
|
|
5229
|
-
reason: `need >= ${t.minRepsForVerdict} reps on both sides (candidate n=${candidate.n}, parent n=${parent.n})`
|
|
5230
|
-
};
|
|
5231
|
-
if (!candidate.stable) return {
|
|
5232
|
-
verdict: "NOISE",
|
|
5233
|
-
medianDelta: candidate.median - parent.median,
|
|
5234
|
-
reason: `candidate unstable (iqr=${candidate.iqr}, stddev=${candidate.stddev.toFixed(2)})`
|
|
5235
|
-
};
|
|
5236
|
-
const medianDelta = candidate.median - parent.median;
|
|
5237
|
-
if (medianDelta > t.keepThreshold) return {
|
|
5238
|
-
verdict: "KEEP",
|
|
5239
|
-
medianDelta,
|
|
5240
|
-
reason: `median +${medianDelta} > +${t.keepThreshold}`
|
|
5241
|
-
};
|
|
5242
|
-
if (medianDelta < -t.regressionThreshold) return {
|
|
5243
|
-
verdict: "REGRESSION",
|
|
5244
|
-
medianDelta,
|
|
5245
|
-
reason: `median ${medianDelta} < -${t.regressionThreshold}`
|
|
5246
|
-
};
|
|
5247
|
-
return {
|
|
5248
|
-
verdict: "NOISE",
|
|
5249
|
-
medianDelta,
|
|
5250
|
-
reason: `median delta ${medianDelta} inside noise band [-${t.regressionThreshold}, +${t.keepThreshold}]`
|
|
5251
|
-
};
|
|
5252
|
-
}
|
|
5253
|
-
/**
|
|
5254
|
-
* Default provenance reader: `git rev-parse HEAD`, the subject line, and the
|
|
5255
|
-
* files changed vs `HEAD~1`. Fail-loud — a tracker that silently logs
|
|
5256
|
-
* `commit: 'unknown'` corrupts the provenance the whole point of the log is to
|
|
5257
|
-
* carry. When the working tree genuinely has no parent commit, pass an override.
|
|
5258
|
-
*/
|
|
5259
|
-
const gitProvenanceReader = () => {
|
|
5260
|
-
const run = (cmd) => execSync(cmd, { encoding: "utf8" }).trim();
|
|
5261
|
-
const commit = run("git rev-parse --short HEAD");
|
|
5262
|
-
const message = run("git log -1 --format=%s");
|
|
5263
|
-
const changedRaw = run("git diff --name-only HEAD~1");
|
|
5264
|
-
return {
|
|
5265
|
-
commit,
|
|
5266
|
-
message,
|
|
5267
|
-
changedFiles: changedRaw.length === 0 ? [] : changedRaw.split("\n").filter(Boolean)
|
|
5268
|
-
};
|
|
5269
|
-
};
|
|
5270
|
-
/** In-memory store — the default when no persistence is wanted (tests, ephemeral
|
|
5271
|
-
* runs). State lives on the instance. */
|
|
5272
|
-
function inMemoryExperimentStore(initial = []) {
|
|
5273
|
-
let state = initial.map((e) => structuredClone(e));
|
|
5274
|
-
return {
|
|
5275
|
-
async load() {
|
|
5276
|
-
return state.map((e) => structuredClone(e));
|
|
5277
|
-
},
|
|
5278
|
-
async save(experiments) {
|
|
5279
|
-
state = experiments.map((e) => structuredClone(e));
|
|
5280
|
-
}
|
|
5281
|
-
};
|
|
5282
|
-
}
|
|
5283
|
-
/** Filesystem store — a single JSON array at `path`, created on first save. */
|
|
5284
|
-
function fileExperimentStore(path) {
|
|
5285
|
-
return {
|
|
5286
|
-
async load() {
|
|
5287
|
-
const fs = await import("node:fs/promises");
|
|
5288
|
-
try {
|
|
5289
|
-
const raw = await fs.readFile(path, "utf8");
|
|
5290
|
-
const parsed = JSON.parse(raw);
|
|
5291
|
-
if (!Array.isArray(parsed)) throw new ValidationError(`experiment-tracker: store at ${path} is not a JSON array`);
|
|
5292
|
-
return parsed;
|
|
5293
|
-
} catch (err) {
|
|
5294
|
-
if (err.code === "ENOENT") return [];
|
|
5295
|
-
throw err;
|
|
5296
|
-
}
|
|
5297
|
-
},
|
|
5298
|
-
async save(experiments) {
|
|
5299
|
-
const fs = await import("node:fs/promises");
|
|
5300
|
-
const pathMod = await import("node:path");
|
|
5301
|
-
await fs.mkdir(pathMod.dirname(path), { recursive: true });
|
|
5302
|
-
await fs.writeFile(path, JSON.stringify(experiments, null, 2), "utf8");
|
|
5303
|
-
}
|
|
5304
|
-
};
|
|
5305
|
-
}
|
|
5306
|
-
/**
|
|
5307
|
-
* Stateful tracker over an `ExperimentStore`. Create an experiment (provenance
|
|
5308
|
-
* is captured once), append reps as they complete (stats + verdict recompute on
|
|
5309
|
-
* every append), and read the log back for a dashboard. All persistence and git
|
|
5310
|
-
* access flow through the injected seams, so the tracker is fully testable
|
|
5311
|
-
* without a repo or disk.
|
|
5312
|
-
*/
|
|
5313
|
-
var ExperimentTracker = class {
|
|
5314
|
-
store;
|
|
5315
|
-
provenanceReader;
|
|
5316
|
-
thresholds;
|
|
5317
|
-
now;
|
|
5318
|
-
constructor(options = {}) {
|
|
5319
|
-
this.store = options.store ?? inMemoryExperimentStore();
|
|
5320
|
-
this.provenanceReader = options.provenanceReader ?? gitProvenanceReader;
|
|
5321
|
-
this.thresholds = resolveThresholds(options.thresholds);
|
|
5322
|
-
this.now = options.now ?? Date.now;
|
|
5323
|
-
}
|
|
5324
|
-
async create(input) {
|
|
5325
|
-
const experiments = await this.store.load();
|
|
5326
|
-
if (experiments.some((e) => e.id === input.id)) throw new ValidationError(`experiment-tracker: experiment id "${input.id}" already exists`);
|
|
5327
|
-
if (input.parentId && !experiments.some((e) => e.id === input.parentId)) throw new ValidationError(`experiment-tracker: parent experiment "${input.parentId}" not found`);
|
|
5328
|
-
const provenance = input.provenance ?? await this.provenanceReader();
|
|
5329
|
-
const experiment = {
|
|
5330
|
-
id: input.id,
|
|
5331
|
-
label: input.label,
|
|
5332
|
-
provenance,
|
|
5333
|
-
parentId: input.parentId,
|
|
5334
|
-
changeSummary: input.changeSummary,
|
|
5335
|
-
reps: [],
|
|
5336
|
-
stats: computeExperimentStats([], this.thresholds),
|
|
5337
|
-
verdict: "ITERATE",
|
|
5338
|
-
createdAt: new Date(this.now()).toISOString()
|
|
5339
|
-
};
|
|
5340
|
-
experiments.push(experiment);
|
|
5341
|
-
await this.store.save(experiments);
|
|
5342
|
-
return structuredClone(experiment);
|
|
5343
|
-
}
|
|
5344
|
-
/** Append a rep (its `rep` index defaults to the current rep count) and
|
|
5345
|
-
* recompute stats + verdict. Returns the updated experiment. */
|
|
5346
|
-
async addRep(experimentId, rep) {
|
|
5347
|
-
const experiments = await this.store.load();
|
|
5348
|
-
const exp = experiments.find((e) => e.id === experimentId);
|
|
5349
|
-
if (!exp) throw new ValidationError(`experiment-tracker: experiment "${experimentId}" not found`);
|
|
5350
|
-
const fullRep = {
|
|
5351
|
-
rep: rep.rep ?? exp.reps.length,
|
|
5352
|
-
score: rep.score,
|
|
5353
|
-
passed: rep.passed,
|
|
5354
|
-
metrics: rep.metrics,
|
|
5355
|
-
timestamp: rep.timestamp ?? new Date(this.now()).toISOString()
|
|
5356
|
-
};
|
|
5357
|
-
exp.reps.push(fullRep);
|
|
5358
|
-
exp.stats = computeExperimentStats(exp.reps, this.thresholds);
|
|
5359
|
-
const parent = exp.parentId ? experiments.find((e) => e.id === exp.parentId) : void 0;
|
|
5360
|
-
exp.verdict = improvementVerdict(exp.stats, parent?.stats ?? null, this.thresholds).verdict;
|
|
5361
|
-
await this.store.save(experiments);
|
|
5362
|
-
return structuredClone(exp);
|
|
5363
|
-
}
|
|
5364
|
-
async get(experimentId) {
|
|
5365
|
-
const found = (await this.store.load()).find((e) => e.id === experimentId);
|
|
5366
|
-
return found ? structuredClone(found) : void 0;
|
|
5367
|
-
}
|
|
5368
|
-
async list() {
|
|
5369
|
-
return this.store.load();
|
|
5370
|
-
}
|
|
5371
|
-
/** Full verdict (not just the enum) for an experiment vs its parent. */
|
|
5372
|
-
async verdictFor(experimentId) {
|
|
5373
|
-
const experiments = await this.store.load();
|
|
5374
|
-
const exp = experiments.find((e) => e.id === experimentId);
|
|
5375
|
-
if (!exp) throw new ValidationError(`experiment-tracker: experiment "${experimentId}" not found`);
|
|
5376
|
-
const parent = exp.parentId ? experiments.find((e) => e.id === exp.parentId) : void 0;
|
|
5377
|
-
return improvementVerdict(exp.stats, parent?.stats ?? null, this.thresholds);
|
|
5378
|
-
}
|
|
5379
|
-
};
|
|
5380
|
-
//#endregion
|
|
5381
4855
|
//#region src/leaderboard.ts
|
|
5382
4856
|
function mean$1(xs) {
|
|
5383
4857
|
return xs.length ? xs.reduce((a, b) => a + b, 0) / xs.length : null;
|
|
@@ -6296,6 +5770,162 @@ function statusAdvanced(key, progression) {
|
|
|
6296
5770
|
};
|
|
6297
5771
|
}
|
|
6298
5772
|
//#endregion
|
|
5773
|
+
//#region src/verification-strategy.ts
|
|
5774
|
+
/**
|
|
5775
|
+
* The family registry. `Record` over the union keeps it exhaustive: adding
|
|
5776
|
+
* a member to `VerificationStrategySource` without a profile here fails to
|
|
5777
|
+
* compile. The failure mode travels with the taxonomy so a reader of a
|
|
5778
|
+
* certification can surface it without this package's docs at hand.
|
|
5779
|
+
*/
|
|
5780
|
+
const VERIFICATION_STRATEGIES = {
|
|
5781
|
+
compile: {
|
|
5782
|
+
determinism: "deterministic",
|
|
5783
|
+
failureMode: "code that compiles is not code that is correct"
|
|
5784
|
+
},
|
|
5785
|
+
test: {
|
|
5786
|
+
determinism: "deterministic",
|
|
5787
|
+
failureMode: "assumes an answer key; certifies nothing outside suite coverage, and a stubbed integration reports green"
|
|
5788
|
+
},
|
|
5789
|
+
schema: {
|
|
5790
|
+
determinism: "deterministic",
|
|
5791
|
+
failureMode: "shape is not meaning; a well-formed wrong answer passes"
|
|
5792
|
+
},
|
|
5793
|
+
sandbox: {
|
|
5794
|
+
determinism: "deterministic",
|
|
5795
|
+
failureMode: "an exit code compresses the run to one bit; a faked success exits 0"
|
|
5796
|
+
},
|
|
5797
|
+
judge: {
|
|
5798
|
+
determinism: "probabilistic",
|
|
5799
|
+
failureMode: "drifts across model versions and is Goodhart-gameable by the graded policy"
|
|
5800
|
+
},
|
|
5801
|
+
composite: {
|
|
5802
|
+
determinism: "inherited",
|
|
5803
|
+
failureMode: "scalar collapse: the blend hides which member carried the score"
|
|
5804
|
+
},
|
|
5805
|
+
"proof-kernel": {
|
|
5806
|
+
determinism: "deterministic",
|
|
5807
|
+
failureMode: "the formalization gap: the kernel certifies the formal statement, never that it matches the informal claim"
|
|
5808
|
+
},
|
|
5809
|
+
invariant: {
|
|
5810
|
+
determinism: "deterministic",
|
|
5811
|
+
failureMode: "weak invariants pass everything; a set uncalibrated by seeded bugs is a rubber stamp"
|
|
5812
|
+
},
|
|
5813
|
+
replication: {
|
|
5814
|
+
determinism: "deterministic",
|
|
5815
|
+
failureMode: "re-runs the method, so it catches drift and nondeterminism, never an error the method itself carries"
|
|
5816
|
+
},
|
|
5817
|
+
agreement: {
|
|
5818
|
+
determinism: "probabilistic",
|
|
5819
|
+
failureMode: "the shared blind spot: derivers with common corpora or priors agree for the same wrong reason"
|
|
5820
|
+
}
|
|
5821
|
+
};
|
|
5822
|
+
/** Every family member, derived from the registry so it cannot drift. */
|
|
5823
|
+
const VERIFICATION_STRATEGY_SOURCES = Object.keys(VERIFICATION_STRATEGIES);
|
|
5824
|
+
//#endregion
|
|
5825
|
+
//#region src/equivalence-check.ts
|
|
5826
|
+
/** A refused equivalence check. `code` names the exact refusal for programmatic handling. */
|
|
5827
|
+
var EquivalenceProtocolError = class extends Error {
|
|
5828
|
+
code;
|
|
5829
|
+
constructor(code, message) {
|
|
5830
|
+
super(message);
|
|
5831
|
+
this.name = "EquivalenceProtocolError";
|
|
5832
|
+
this.code = code;
|
|
5833
|
+
}
|
|
5834
|
+
};
|
|
5835
|
+
/**
|
|
5836
|
+
* Validate and freeze a two-arm blind equivalence check spec.
|
|
5837
|
+
*
|
|
5838
|
+
* The literal types already refuse a wide design at compile time; the
|
|
5839
|
+
* runtime checks hold the same line for untyped callers. There is no
|
|
5840
|
+
* escape hatch: `arms: 3` or `blind: false` throws, never downgrades.
|
|
5841
|
+
*/
|
|
5842
|
+
function defineEquivalenceCheck(spec) {
|
|
5843
|
+
if (spec.arms !== 2) throw new EquivalenceProtocolError("arm-count", `equivalence check requires exactly 2 arms, received ${String(spec.arms)}`);
|
|
5844
|
+
if (spec.blind !== true) throw new EquivalenceProtocolError("not-blind", "equivalence check requires blind: true — a non-blind run is not a weaker check, it is no check");
|
|
5845
|
+
if (typeof spec.artifact !== "string" || spec.artifact.trim() === "") throw new EquivalenceProtocolError("empty-field", "spec.artifact must identify the claim under verification");
|
|
5846
|
+
if (!VERIFICATION_STRATEGY_SOURCES.includes(spec.source)) throw new EquivalenceProtocolError("unknown-source", `spec.source '${String(spec.source)}' is not a verification-strategy member`);
|
|
5847
|
+
return Object.freeze({ spec: Object.freeze({ ...spec }) });
|
|
5848
|
+
}
|
|
5849
|
+
/**
|
|
5850
|
+
* Assemble the equivalence record, refusing every asymmetry.
|
|
5851
|
+
*
|
|
5852
|
+
* Refusals (each throws `EquivalenceProtocolError`):
|
|
5853
|
+
* - an arm whose `blindness.toOtherArms` is false — it saw the other
|
|
5854
|
+
* statement, so nothing was independently derived;
|
|
5855
|
+
* - an arm whose `blindness.toOutcome` is false — it could steer its
|
|
5856
|
+
* statement toward (or away from) the known result;
|
|
5857
|
+
* - duplicate arm ids, empty statements, empty derivations;
|
|
5858
|
+
* - an obligation whose fields contradict its status (see
|
|
5859
|
+
* `EquivalenceObligation`).
|
|
5860
|
+
*/
|
|
5861
|
+
function buildEquivalenceRecord(definition, arms, obligation) {
|
|
5862
|
+
assertArms(arms);
|
|
5863
|
+
assertObligation(obligation);
|
|
5864
|
+
return Object.freeze({
|
|
5865
|
+
spec: definition.spec,
|
|
5866
|
+
arms: Object.freeze([Object.freeze({ ...arms[0] }), Object.freeze({ ...arms[1] })]),
|
|
5867
|
+
obligation: Object.freeze({ ...obligation })
|
|
5868
|
+
});
|
|
5869
|
+
}
|
|
5870
|
+
/**
|
|
5871
|
+
* Discharge the obligation through an injected checker and return the
|
|
5872
|
+
* record.
|
|
5873
|
+
*
|
|
5874
|
+
* Order matters: every arm refusal fires BEFORE the checker runs — an
|
|
5875
|
+
* invalid check must not spend. A checker whose `strategy` differs from
|
|
5876
|
+
* `spec.source` is refused for the same reason: a judge cannot silently
|
|
5877
|
+
* discharge a proof-kernel obligation.
|
|
5878
|
+
*
|
|
5879
|
+
* A checker failure (`succeeded: false`) is not thrown: it becomes an
|
|
5880
|
+
* `'unresolved'` obligation carrying the full error text, which is the
|
|
5881
|
+
* honest record of an undischarged check.
|
|
5882
|
+
*/
|
|
5883
|
+
async function runEquivalenceCheck(definition, arms, checker) {
|
|
5884
|
+
assertArms(arms);
|
|
5885
|
+
if (checker.strategy !== definition.spec.source) throw new EquivalenceProtocolError("checker-strategy-mismatch", `spec.source is '${definition.spec.source}' but the bound checker declares '${checker.strategy}'`);
|
|
5886
|
+
const outcome = await checker.check({
|
|
5887
|
+
artifact: definition.spec.artifact,
|
|
5888
|
+
statements: [arms[0].statement, arms[1].statement]
|
|
5889
|
+
});
|
|
5890
|
+
if (!outcome.succeeded) return buildEquivalenceRecord(definition, arms, {
|
|
5891
|
+
status: "unresolved",
|
|
5892
|
+
unresolvedReason: outcome.error,
|
|
5893
|
+
checker: checker.identity
|
|
5894
|
+
});
|
|
5895
|
+
const { status, separatingWitness, evidenceDigest } = outcome.value;
|
|
5896
|
+
return buildEquivalenceRecord(definition, arms, {
|
|
5897
|
+
status,
|
|
5898
|
+
...separatingWitness === void 0 ? {} : { separatingWitness },
|
|
5899
|
+
checker: checker.identity,
|
|
5900
|
+
evidenceDigest
|
|
5901
|
+
});
|
|
5902
|
+
}
|
|
5903
|
+
function assertArms(arms) {
|
|
5904
|
+
if (arms.length !== 2) throw new EquivalenceProtocolError("arm-count", `equivalence record requires exactly 2 arms, received ${arms.length}`);
|
|
5905
|
+
if (arms[0].armId === arms[1].armId) throw new EquivalenceProtocolError("duplicate-arm-id", `both arms declare armId '${arms[0].armId}' — two labels for one derivation is one arm`);
|
|
5906
|
+
for (const arm of arms) {
|
|
5907
|
+
if (arm.statement.trim() === "") throw new EquivalenceProtocolError("empty-field", `arm '${arm.armId}' committed an empty statement`);
|
|
5908
|
+
if (arm.derivedFrom.trim() === "") throw new EquivalenceProtocolError("empty-field", `arm '${arm.armId}' declares no derivation provenance`);
|
|
5909
|
+
if (arm.blindness.toOtherArms !== true) throw new EquivalenceProtocolError("arm-saw-other", `arm '${arm.armId}' saw another arm's statement — the derivation is not independent and the check is invalid`);
|
|
5910
|
+
if (arm.blindness.toOutcome !== true) throw new EquivalenceProtocolError("arm-saw-outcome", `arm '${arm.armId}' saw the outcome before committing — the check is invalid`);
|
|
5911
|
+
}
|
|
5912
|
+
}
|
|
5913
|
+
function assertObligation(obligation) {
|
|
5914
|
+
const { status, separatingWitness, unresolvedReason, evidenceDigest } = obligation;
|
|
5915
|
+
if (status === "refuted-with-separating-witness") {
|
|
5916
|
+
if (typeof separatingWitness !== "string" || separatingWitness.trim() === "") throw new EquivalenceProtocolError("witness-missing", "a refuted equivalence must carry the separating witness — a refutation without one is an assertion");
|
|
5917
|
+
if (typeof evidenceDigest !== "string" || evidenceDigest.trim() === "") throw new EquivalenceProtocolError("evidence-missing", "a refuted equivalence must carry its evidence digest");
|
|
5918
|
+
return;
|
|
5919
|
+
}
|
|
5920
|
+
if (status === "proved") {
|
|
5921
|
+
if (separatingWitness !== void 0) throw new EquivalenceProtocolError("witness-on-proved", "a proved equivalence cannot carry a separating witness — the two claims contradict");
|
|
5922
|
+
if (typeof evidenceDigest !== "string" || evidenceDigest.trim() === "") throw new EquivalenceProtocolError("evidence-missing", "a proved equivalence must carry its evidence digest");
|
|
5923
|
+
return;
|
|
5924
|
+
}
|
|
5925
|
+
if (separatingWitness !== void 0) throw new EquivalenceProtocolError("witness-on-unresolved", "an unresolved obligation cannot carry a separating witness — a witness in hand is a refutation");
|
|
5926
|
+
if (typeof unresolvedReason !== "string" || unresolvedReason.trim() === "") throw new EquivalenceProtocolError("reason-missing", "an unresolved obligation must say why — 'unresolved' with no reason erases the diagnostic");
|
|
5927
|
+
}
|
|
5928
|
+
//#endregion
|
|
6299
5929
|
//#region src/ui-finding.ts
|
|
6300
5930
|
/** Frozen tuple of lenses for validation + iteration. */
|
|
6301
5931
|
const UI_LENSES = [
|
|
@@ -7496,126 +7126,6 @@ async function promptBisect(options) {
|
|
|
7496
7126
|
};
|
|
7497
7127
|
}
|
|
7498
7128
|
//#endregion
|
|
7499
|
-
//#region src/counterfactual.ts
|
|
7500
|
-
/**
|
|
7501
|
-
* Counterfactual replay — "what would have happened if we'd changed
|
|
7502
|
-
* exactly one thing at turn N?"
|
|
7503
|
-
*
|
|
7504
|
-
* The framework does NOT drive the agent — it sets up the replay
|
|
7505
|
-
* context (prior spans, prior state, mutation spec) and records the
|
|
7506
|
-
* resulting divergence. Consumers supply an `executeFrom(ctx)` callback
|
|
7507
|
-
* that runs their agent starting from turn N with the mutation applied.
|
|
7508
|
-
*
|
|
7509
|
-
* Counterfactual runs are recorded as a new Run with `layer='meta'` and
|
|
7510
|
-
* `parentRunId = originalRunId`, so downstream diff + correlation
|
|
7511
|
-
* pipelines see them natively.
|
|
7512
|
-
*/
|
|
7513
|
-
async function runCounterfactual(store, originalRunId, mutation, runner) {
|
|
7514
|
-
const originalRun = await store.getRun(originalRunId);
|
|
7515
|
-
if (!originalRun) throw new NotFoundError(`counterfactual: run ${originalRunId} not found`);
|
|
7516
|
-
const trajectory = await buildTrajectory(store, originalRunId);
|
|
7517
|
-
if (mutation.at < 0 || mutation.at >= trajectory.steps.length) throw new ValidationError(`counterfactual: mutation.at=${mutation.at} out of range [0, ${trajectory.steps.length})`);
|
|
7518
|
-
const targetStep = trajectory.steps[mutation.at];
|
|
7519
|
-
const mutatedStep = applyMutation(targetStep, mutation);
|
|
7520
|
-
const cfEmitter = new TraceEmitter(store);
|
|
7521
|
-
await cfEmitter.startRun({
|
|
7522
|
-
scenarioId: originalRun.scenarioId,
|
|
7523
|
-
variantId: originalRun.variantId ? `${originalRun.variantId}+cf:${mutation.kind}@${mutation.at}` : `cf:${mutation.kind}@${mutation.at}`,
|
|
7524
|
-
projectId: originalRun.projectId,
|
|
7525
|
-
parentRunId: originalRunId,
|
|
7526
|
-
layer: "meta",
|
|
7527
|
-
tags: {
|
|
7528
|
-
counterfactual: "true",
|
|
7529
|
-
mutationKind: mutation.kind,
|
|
7530
|
-
mutationAt: String(mutation.at)
|
|
7531
|
-
}
|
|
7532
|
-
});
|
|
7533
|
-
await runner.executeFrom({
|
|
7534
|
-
originalRunId,
|
|
7535
|
-
originalTrajectory: trajectory,
|
|
7536
|
-
prefix: trajectory.steps.slice(0, mutation.at),
|
|
7537
|
-
mutation,
|
|
7538
|
-
mutatedStep
|
|
7539
|
-
}, cfEmitter);
|
|
7540
|
-
const counterfactual = await store.getRun(cfEmitter.runId);
|
|
7541
|
-
const delta = {
|
|
7542
|
-
originalOutcomeScore: originalRun.outcome?.score ?? null,
|
|
7543
|
-
counterfactualOutcomeScore: counterfactual?.outcome?.score ?? null,
|
|
7544
|
-
deltaScore: originalRun.outcome?.score !== void 0 && counterfactual?.outcome?.score !== void 0 ? counterfactual.outcome.score - originalRun.outcome.score : null
|
|
7545
|
-
};
|
|
7546
|
-
return {
|
|
7547
|
-
counterfactualRunId: cfEmitter.runId,
|
|
7548
|
-
originalRunId,
|
|
7549
|
-
mutation,
|
|
7550
|
-
delta
|
|
7551
|
-
};
|
|
7552
|
-
}
|
|
7553
|
-
function applyMutation(step, mutation) {
|
|
7554
|
-
if (mutation.kind === "swap-model" && step.span.kind === "llm") {
|
|
7555
|
-
const llm = step.span;
|
|
7556
|
-
return {
|
|
7557
|
-
...step,
|
|
7558
|
-
span: {
|
|
7559
|
-
...llm,
|
|
7560
|
-
model: mutation.newModel
|
|
7561
|
-
}
|
|
7562
|
-
};
|
|
7563
|
-
}
|
|
7564
|
-
if (mutation.kind === "swap-tool-result" && step.span.kind === "tool") {
|
|
7565
|
-
const tool = step.span;
|
|
7566
|
-
return {
|
|
7567
|
-
...step,
|
|
7568
|
-
span: {
|
|
7569
|
-
...tool,
|
|
7570
|
-
result: mutation.newResult
|
|
7571
|
-
}
|
|
7572
|
-
};
|
|
7573
|
-
}
|
|
7574
|
-
if (mutation.kind === "inject-system-message" && step.span.kind === "llm") {
|
|
7575
|
-
const llm = step.span;
|
|
7576
|
-
return {
|
|
7577
|
-
...step,
|
|
7578
|
-
span: {
|
|
7579
|
-
...llm,
|
|
7580
|
-
messages: [{
|
|
7581
|
-
role: "system",
|
|
7582
|
-
content: mutation.content
|
|
7583
|
-
}, ...llm.messages]
|
|
7584
|
-
}
|
|
7585
|
-
};
|
|
7586
|
-
}
|
|
7587
|
-
if (mutation.kind === "custom") return mutation.apply(step);
|
|
7588
|
-
return step;
|
|
7589
|
-
}
|
|
7590
|
-
/**
|
|
7591
|
-
* Aggregate a batch of counterfactuals into a simple attribution table:
|
|
7592
|
-
* which mutation kinds move outcomes most? (Useful when you run a grid
|
|
7593
|
-
* over the same trajectory — swap-model at every llm span, swap-tool
|
|
7594
|
-
* at every tool span — and want a ranked summary.)
|
|
7595
|
-
*/
|
|
7596
|
-
function attributeCounterfactuals(results) {
|
|
7597
|
-
const grouped = /* @__PURE__ */ new Map();
|
|
7598
|
-
for (const r of results) {
|
|
7599
|
-
const arr = grouped.get(r.mutation.kind) ?? [];
|
|
7600
|
-
arr.push(r);
|
|
7601
|
-
grouped.set(r.mutation.kind, arr);
|
|
7602
|
-
}
|
|
7603
|
-
const out = [];
|
|
7604
|
-
for (const [kind, items] of grouped) {
|
|
7605
|
-
const deltas = items.map((i) => i.delta.deltaScore).filter((d) => typeof d === "number");
|
|
7606
|
-
if (deltas.length === 0) continue;
|
|
7607
|
-
const meanAbs = deltas.reduce((a, b) => a + Math.abs(b), 0) / deltas.length;
|
|
7608
|
-
const meanSigned = deltas.reduce((a, b) => a + b, 0) / deltas.length;
|
|
7609
|
-
out.push({
|
|
7610
|
-
mutationKind: kind,
|
|
7611
|
-
n: deltas.length,
|
|
7612
|
-
meanAbsDelta: meanAbs,
|
|
7613
|
-
meanSignedDelta: meanSigned
|
|
7614
|
-
});
|
|
7615
|
-
}
|
|
7616
|
-
return out.sort((a, b) => b.meanAbsDelta - a.meanAbsDelta);
|
|
7617
|
-
}
|
|
7618
|
-
//#endregion
|
|
7619
7129
|
//#region src/cross-trace-diff.ts
|
|
7620
7130
|
async function crossTraceDiff(store, runA, runB, options = {}) {
|
|
7621
7131
|
const [a, b] = await Promise.all([buildTrajectory(store, runA), buildTrajectory(store, runB)]);
|
|
@@ -11470,10 +10980,20 @@ function attachCostToReport(report, ledger) {
|
|
|
11470
10980
|
/**
|
|
11471
10981
|
* Tier presets — plain data, swap or spread freely.
|
|
11472
10982
|
*
|
|
11473
|
-
*
|
|
11474
|
-
*
|
|
11475
|
-
*
|
|
11476
|
-
* `
|
|
10983
|
+
* A preset names REQUESTED ids, and a requested id is NOT a guarantee of
|
|
10984
|
+
* family, of provider, or of liveness. A routing gateway can accept any id
|
|
10985
|
+
* below and answer from a different model on HTTP 200, with only the response
|
|
10986
|
+
* body's `model` field betraying the swap. The served assertion is the
|
|
10987
|
+
* guarantee: gate a run on `assertModelsServed({ probe: true })` and assert
|
|
10988
|
+
* `assertServedModel` per call. `assertCrossFamily` over these ids proves the
|
|
10989
|
+
* configuration is diverse; only `assertCrossFamilyServed` over the ids that
|
|
10990
|
+
* ANSWERED proves the run was.
|
|
10991
|
+
*
|
|
10992
|
+
* `economy` names ids a live router probe answered from the provider their
|
|
10993
|
+
* name implies, so the judge trio spanned three provider families (deepseek /
|
|
10994
|
+
* zhipu / google) as configured and, at that probe, as served. A preset is
|
|
10995
|
+
* only as good as its last probe: ids go dead and start resolving elsewhere
|
|
10996
|
+
* without notice, so re-probe rather than trusting this list.
|
|
11477
10997
|
*
|
|
11478
10998
|
* `frontier` is deliberately EMPTY: entitled frontier ids vary per router
|
|
11479
10999
|
* account, and a hardcoded claude/gpt-5 id 401s on keys that lack it. Supply
|
|
@@ -11482,14 +11002,14 @@ function attachCostToReport(report, ledger) {
|
|
|
11482
11002
|
*/
|
|
11483
11003
|
const seatPresets = {
|
|
11484
11004
|
economy: {
|
|
11485
|
-
worker: "
|
|
11005
|
+
worker: "zai/glm-5.2",
|
|
11486
11006
|
judges: [
|
|
11487
|
-
"kimi-k2.6",
|
|
11488
11007
|
"deepseek-v4-pro",
|
|
11489
|
-
"
|
|
11008
|
+
"zai/glm-5.2",
|
|
11009
|
+
"google/gemini-2.5-flash"
|
|
11490
11010
|
],
|
|
11491
|
-
analyst: "
|
|
11492
|
-
reflection: "
|
|
11011
|
+
analyst: "zai/glm-5.2",
|
|
11012
|
+
reflection: "zai/glm-5.2",
|
|
11493
11013
|
verifier: "deepseek-v4-pro"
|
|
11494
11014
|
},
|
|
11495
11015
|
frontier: {}
|
|
@@ -12464,6 +11984,6 @@ function assertProductBenchmarkRun(runDir) {
|
|
|
12464
11984
|
return report;
|
|
12465
11985
|
}
|
|
12466
11986
|
//#endregion
|
|
12467
|
-
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CONTROL_INTEGRITY_ANALYST, CallbackResearcher, CaptureIntegrityError, ConfigError, ControlIntegrityAnalyst, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExactAnalystRunExecutionError, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_BRIEFS, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LimitExceededError, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_INTEGRITY_SCHEMA, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYSIS_LIMITS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TOOL_NAMESPACE, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, ToolTraceMissingError, TraceAnalysisLimitError, TraceAnalysisStoreContractError, TraceAnalysisValidationError, TraceContractBuilder, TraceEmitter, TraceFileMalformedError, TraceFileMissingError, TraceFileTooLargeError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, WORKER_DRIVER_DOCTRINE, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunIntegrity, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertExactRegistryRunOpts, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalysisToolDescriptors, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDefaultReviewer, createDspyRlmTraceEngine, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalyst, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, defineTraceAnalyst, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, emitControlIntegrityFindings, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isRuntimeSupervisorRunDir, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpTextToTraceAnalysisStore, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, readRuntimeSupervisorRun, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resetDeprecationWarnings, resolveModelPricing, resolveSeat, resolveTraceAnalystLimits, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runTraceAnalyst, runsForScenario, runtimeSupervisorRunReader, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, toolSpansToTraceAnalysisStore, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, warnDeprecatedOnce, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
11987
|
+
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CONTROL_INTEGRITY_ANALYST, CallbackResearcher, CaptureIntegrityError, ConfigError, ControlIntegrityAnalyst, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, EvalTraceStore, ExactAnalystRunExecutionError, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_BRIEFS, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LimitExceededError, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelSubstitutionError, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PROBE_MAX_TOKENS, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_INTEGRITY_SCHEMA, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, ServedCrossFamilyError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYSIS_LIMITS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TOOL_NAMESPACE, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, ToolTraceMissingError, TraceAnalysisLimitError, TraceAnalysisStoreContractError, TraceAnalysisValidationError, TraceContractBuilder, TraceEmitter, TraceFileMalformedError, TraceFileMissingError, TraceFileTooLargeError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, WORKER_DRIVER_DOCTRINE, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunIntegrity, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertExactRegistryRunOpts, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildEquivalenceRecord, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalysisToolDescriptors, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkServedModel, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDefaultReviewer, createDspyRlmTraceEngine, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalyst, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, defineEquivalenceCheck, defineTraceAnalyst, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, emitControlIntegrityFindings, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isRuntimeSupervisorRunDir, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeModelId, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpTextToTraceAnalysisStore, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, readRuntimeSupervisorRun, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resetDeprecationWarnings, resolveModelPricing, resolveSeat, resolveTraceAnalystLimits, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEquivalenceCheck, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runTraceAnalyst, runsForScenario, runtimeSupervisorRunReader, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, servedModelAcceptable, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, toolSpansToTraceAnalysisStore, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, warnDeprecatedOnce, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
12468
11988
|
|
|
12469
11989
|
//# sourceMappingURL=index.js.map
|