@tangle-network/agent-eval 0.144.5 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +471 -89
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +8 -6
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
- package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
- package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +11 -10
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
- package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
- package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
- package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
- package/dist/errors-D-LKuDhb.js.map +1 -1
- package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
- package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
- package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
- package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
- package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +252 -451
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +271 -751
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
- package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
- package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
- package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +2 -2
- package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
- package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +19 -9
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
- package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
- package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
- package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
- package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
- package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
- package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
- package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/building-doctrine.md +15 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +21 -4
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
- package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
- package/dist/campaign-ClpnD7Ug.js.map +0 -1
- package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
- package/dist/index-CpxZSlB7.d.ts.map +0 -1
- package/dist/index-CsuAo2-J.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-D3EoChAU.js.map +0 -1
- package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
- package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
- package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
- package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
- package/dist/types-BjMFz88h.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { r as CaptureIntegrityError, t as AgentEvalError } from "./errors-
|
|
2
|
-
import { b as CustomTokenPricing, c as CostLedgerHandle, f as CostLedgerSummary, g as CostReceiptInput, x as MaximumCharge } from "./cost-ledger-
|
|
1
|
+
import { r as CaptureIntegrityError, t as AgentEvalError } from "./errors-CKPfb2aH.js";
|
|
2
|
+
import { b as CustomTokenPricing, c as CostLedgerHandle, f as CostLedgerSummary, g as CostReceiptInput, x as MaximumCharge } from "./cost-ledger-Bv_e8XHY.js";
|
|
3
3
|
//#region src/trace/raw-provider-sink.d.ts
|
|
4
4
|
/**
|
|
5
5
|
* RawProviderSink — first-class persistence for the actual HTTP-level
|
|
@@ -132,6 +132,186 @@ declare class FileSystemRawProviderSink implements RawProviderSink {
|
|
|
132
132
|
*/
|
|
133
133
|
declare function providerFromBaseUrl(baseUrl: string): string;
|
|
134
134
|
//#endregion
|
|
135
|
+
//#region src/judge-families.d.ts
|
|
136
|
+
/**
|
|
137
|
+
* Judge model-family classification + cross-family enforcement.
|
|
138
|
+
*
|
|
139
|
+
* A judge ensemble built entirely from one provider family shares that
|
|
140
|
+
* family's blind spots and self-preference — its "agreement" is correlated
|
|
141
|
+
* bias, not independent signal. `assertCrossFamily` makes the consumer prove
|
|
142
|
+
* the ensemble spans ≥2 families; `judgeFamily` is the single regex map that
|
|
143
|
+
* replaces the per-consumer copies (tax/legal/creative/gtm each ship one).
|
|
144
|
+
*/
|
|
145
|
+
/** Provider family a model belongs to. `unknown` when no rule matches. */
|
|
146
|
+
type JudgeFamily = 'anthropic' | 'openai' | 'google' | 'meta' | 'mistral' | 'deepseek' | 'xai' | 'qwen' | 'cohere' | 'amazon' | 'moonshot' | 'zhipu' | 'unknown';
|
|
147
|
+
/**
|
|
148
|
+
* Classify a model id into its provider family. Strips a `@snapshot` suffix
|
|
149
|
+
* and prefers an explicit `provider/...` prefix; otherwise matches the model
|
|
150
|
+
* name. Returns `unknown` when nothing matches (callers decide whether that's
|
|
151
|
+
* acceptable — `assertCrossFamily` counts it as its own family).
|
|
152
|
+
*/
|
|
153
|
+
declare function judgeFamily(modelId: string): JudgeFamily;
|
|
154
|
+
interface AssertCrossFamilyOptions {
|
|
155
|
+
/** Minimum number of distinct families the ensemble must span. Default 2. */
|
|
156
|
+
minFamilies?: number;
|
|
157
|
+
/** When false (default), `unknown`-family models do NOT count toward the
|
|
158
|
+
* family total — an ensemble of all-unclassifiable models is not provably
|
|
159
|
+
* cross-family. Set true to count `unknown` as one shared family. */
|
|
160
|
+
allowUnknown?: boolean;
|
|
161
|
+
}
|
|
162
|
+
declare class CrossFamilyError extends Error {
|
|
163
|
+
readonly families: JudgeFamily[];
|
|
164
|
+
readonly models: string[];
|
|
165
|
+
constructor(message: string, families: JudgeFamily[], models: string[]);
|
|
166
|
+
}
|
|
167
|
+
/**
|
|
168
|
+
* Throw unless the judge models span at least `minFamilies` distinct provider
|
|
169
|
+
* families. Pass the model ids backing your judge ensemble. Fail-loud by
|
|
170
|
+
* design — a correlated single-family ensemble silently inflates agreement.
|
|
171
|
+
*
|
|
172
|
+
* Scope: this reads the ids you REQUEST. It proves the panel was configured
|
|
173
|
+
* across families; it cannot prove the panel RAN across families, because a
|
|
174
|
+
* routing gateway may answer several different ids from one provider. Where
|
|
175
|
+
* the diversity claim is load-bearing (a published leaderboard, a
|
|
176
|
+
* certification, a non-self-judging exclusion), assert on the ids the
|
|
177
|
+
* provider echoed instead: `assertCrossFamilyServed` in
|
|
178
|
+
* ./integrity/served-model.
|
|
179
|
+
*/
|
|
180
|
+
declare function assertCrossFamily(models: string[], opts?: AssertCrossFamilyOptions): JudgeFamily[];
|
|
181
|
+
//#endregion
|
|
182
|
+
//#region src/integrity/served-model.d.ts
|
|
183
|
+
/**
|
|
184
|
+
* Output-token budget a liveness probe must grant a model.
|
|
185
|
+
*
|
|
186
|
+
* Identity is only readable off a response the provider actually produced. A
|
|
187
|
+
* reasoning model spends budget on hidden reasoning tokens before it emits a
|
|
188
|
+
* single visible one, so a cap of a few tokens makes a HEALTHY deepseek/glm
|
|
189
|
+
* model fail with `reasoning_budget_exhausted` — it names no model, its
|
|
190
|
+
* identity reads as `unreported`, and a preflight scores it DEAD. 64 clears
|
|
191
|
+
* that floor. Every probe in this package reads this constant, so two probes
|
|
192
|
+
* cannot reach two different answers about the same router.
|
|
193
|
+
*
|
|
194
|
+
* Cost: a probe spends at most `PROBE_MAX_TOKENS` output tokens per model,
|
|
195
|
+
* plus whatever reasoning tokens a reasoning model bills — roughly 400 output
|
|
196
|
+
* tokens for a six-model preflight. That is fractions of a cent, and far
|
|
197
|
+
* cheaper than a campaign that runs on a model nobody proved was alive.
|
|
198
|
+
*/
|
|
199
|
+
declare const PROBE_MAX_TOKENS = 64;
|
|
200
|
+
/** How a served id relates to the id that was requested. */
|
|
201
|
+
type ServedModelVerdict =
|
|
202
|
+
/** Byte-identical after normalisation — the requested model answered. */
|
|
203
|
+
'exact' |
|
|
204
|
+
/** Same model, different spelling (provider prefix, snapshot, tier suffix). */
|
|
205
|
+
'alias' |
|
|
206
|
+
/** A different model of the SAME provider family answered. */
|
|
207
|
+
'substituted-within-family' |
|
|
208
|
+
/** A different provider's model answered. */
|
|
209
|
+
'substituted-cross-family' |
|
|
210
|
+
/** The response carried no model id — identity is unproven either way. */
|
|
211
|
+
'unreported';
|
|
212
|
+
interface ServedModelCheck {
|
|
213
|
+
/** The id the caller asked for. */
|
|
214
|
+
requested: string;
|
|
215
|
+
/** The id echoed on the response; `null` when the response omitted it. */
|
|
216
|
+
served: string | null;
|
|
217
|
+
requestedFamily: JudgeFamily;
|
|
218
|
+
/** `null` when `served` is null. */
|
|
219
|
+
servedFamily: JudgeFamily | null;
|
|
220
|
+
verdict: ServedModelVerdict;
|
|
221
|
+
/** True for every verdict except `exact` and `alias`. */
|
|
222
|
+
substituted: boolean;
|
|
223
|
+
}
|
|
224
|
+
/**
|
|
225
|
+
* Reduce a model id to its comparable core: lowercase, no surrounding space,
|
|
226
|
+
* no `provider/` prefix, no `@snapshot` / `:batch` / `:free` tier suffix, no
|
|
227
|
+
* trailing build date, and `.`/`_` folded to `-` so one version is spelled one
|
|
228
|
+
* way.
|
|
229
|
+
*
|
|
230
|
+
* Dropping the build date is what makes snapshot resolution legible as the
|
|
231
|
+
* non-event it is: a router answering `gpt-4o-mini` with
|
|
232
|
+
* `gpt-4o-mini-2024-07-18` pinned a floating alias to a reproducible build —
|
|
233
|
+
* the same model, which is the behaviour we want. Only routing decoration is
|
|
234
|
+
* stripped; version digits are load-bearing, so `deepseek-v3.2` and
|
|
235
|
+
* `deepseek-v4-flash` stay distinct, and comparison is EXACT equality rather
|
|
236
|
+
* than a prefix test (a prefix rule would accept `gpt-5` → `gpt-5-mini`, a
|
|
237
|
+
* silent downgrade wearing the right vendor name).
|
|
238
|
+
*/
|
|
239
|
+
declare function normalizeModelId(modelId: string): string;
|
|
240
|
+
/**
|
|
241
|
+
* Classify one requested/served pair. Pure — no I/O — so it is safe inside
|
|
242
|
+
* response handlers, reducers, and CI gates.
|
|
243
|
+
*
|
|
244
|
+
* `served` is the id echoed by the provider (OpenAI-compatible bodies put it
|
|
245
|
+
* at `model`). `null`/`undefined` means the body omitted it; that is
|
|
246
|
+
* `unreported`, NOT a pass — a provider that does not name what answered has
|
|
247
|
+
* not proven identity, and a transport that drops the field must not read as
|
|
248
|
+
* agreement.
|
|
249
|
+
*/
|
|
250
|
+
declare function checkServedModel(requested: string, served: string | null | undefined): ServedModelCheck;
|
|
251
|
+
declare class ModelSubstitutionError extends AgentEvalError {
|
|
252
|
+
readonly checks: ReadonlyArray<ServedModelCheck>;
|
|
253
|
+
constructor(message: string, checks: ReadonlyArray<ServedModelCheck>);
|
|
254
|
+
}
|
|
255
|
+
interface AssertServedModelOptions {
|
|
256
|
+
/**
|
|
257
|
+
* Accept a different model of the same provider family (e.g. requested
|
|
258
|
+
* `deepseek-v3.2`, served `deepseek-v4-flash`). Default false. Setting this
|
|
259
|
+
* keeps family-level claims valid and forfeits per-model claims.
|
|
260
|
+
*/
|
|
261
|
+
allowWithinFamily?: boolean;
|
|
262
|
+
/**
|
|
263
|
+
* Accept a response that carried no model id. Default false — an
|
|
264
|
+
* unidentified response cannot support a per-model or per-family claim.
|
|
265
|
+
*/
|
|
266
|
+
allowUnreported?: boolean;
|
|
267
|
+
/** Prefixed to the thrown message, e.g. the judge or campaign cell name. */
|
|
268
|
+
context?: string;
|
|
269
|
+
}
|
|
270
|
+
/**
|
|
271
|
+
* The one place the accept/reject policy lives, so a caller that reports
|
|
272
|
+
* substitution (a preflight table, a run record) and a caller that throws on it
|
|
273
|
+
* can never drift apart. A cross-family substitution is never acceptable.
|
|
274
|
+
*/
|
|
275
|
+
declare function servedModelAcceptable(check: ServedModelCheck, opts?: AssertServedModelOptions): boolean;
|
|
276
|
+
/**
|
|
277
|
+
* Throw `ModelSubstitutionError` unless the served id is the requested model.
|
|
278
|
+
* Returns the check on success so callers can record the served id alongside
|
|
279
|
+
* the result.
|
|
280
|
+
*/
|
|
281
|
+
declare function assertServedModel(requested: string, served: string | null | undefined, opts?: AssertServedModelOptions): ServedModelCheck;
|
|
282
|
+
/**
|
|
283
|
+
* Batch form: check every pair and throw naming EVERY substitution, so one
|
|
284
|
+
* failure does not hide the rest. Returns all checks on success.
|
|
285
|
+
*/
|
|
286
|
+
declare function assertServedModels(pairs: ReadonlyArray<{
|
|
287
|
+
requested: string;
|
|
288
|
+
served: string | null | undefined;
|
|
289
|
+
}>, opts?: AssertServedModelOptions): ServedModelCheck[];
|
|
290
|
+
interface AssertCrossFamilyServedOptions extends AssertServedModelOptions {
|
|
291
|
+
/** Minimum distinct SERVED families required. Default 2. */
|
|
292
|
+
minFamilies?: number;
|
|
293
|
+
/** Count `unknown`-family served ids toward the total. Default false. */
|
|
294
|
+
allowUnknown?: boolean;
|
|
295
|
+
}
|
|
296
|
+
declare class ServedCrossFamilyError extends AgentEvalError {
|
|
297
|
+
readonly families: JudgeFamily[];
|
|
298
|
+
readonly checks: ReadonlyArray<ServedModelCheck>;
|
|
299
|
+
constructor(message: string, families: JudgeFamily[], checks: ReadonlyArray<ServedModelCheck>);
|
|
300
|
+
}
|
|
301
|
+
/**
|
|
302
|
+
* Family-diversity rule over the models that actually ANSWERED.
|
|
303
|
+
*
|
|
304
|
+
* `assertCrossFamily` (../judge-families) reads the requested ids and so
|
|
305
|
+
* cannot see a gateway that answers three "different" requests from one
|
|
306
|
+
* provider. This one asserts no substitution first, then counts families from
|
|
307
|
+
* the served ids — a panel that collapsed to one family under the hood fails
|
|
308
|
+
* here even though its request list looked diverse.
|
|
309
|
+
*/
|
|
310
|
+
declare function assertCrossFamilyServed(pairs: ReadonlyArray<{
|
|
311
|
+
requested: string;
|
|
312
|
+
served: string | null | undefined;
|
|
313
|
+
}>, opts?: AssertCrossFamilyServedOptions): JudgeFamily[];
|
|
314
|
+
//#endregion
|
|
135
315
|
//#region src/llm-client.d.ts
|
|
136
316
|
interface LlmMessage {
|
|
137
317
|
role: 'system' | 'user' | 'assistant';
|
|
@@ -193,8 +373,27 @@ interface LlmCallResult {
|
|
|
193
373
|
* caller-supplied token pricing. `null` when neither is available.
|
|
194
374
|
*/
|
|
195
375
|
costUsd: number | null;
|
|
196
|
-
/**
|
|
376
|
+
/**
|
|
377
|
+
* Model id used for attribution (cost, pricing, log lines). The response's
|
|
378
|
+
* echoed id when the provider sent one, else the requested id.
|
|
379
|
+
*
|
|
380
|
+
* NOT evidence of which model answered — read `servedModel` for that. A
|
|
381
|
+
* provider that omits `model` makes this equal to the request, which is
|
|
382
|
+
* exactly the case an identity check must be able to distinguish.
|
|
383
|
+
*/
|
|
197
384
|
model: string;
|
|
385
|
+
/**
|
|
386
|
+
* The model id the provider echoed on the response, verbatim; `null` when
|
|
387
|
+
* the body carried none. This is the only field that can witness a gateway
|
|
388
|
+
* substituting a different model for the one requested — compare it with
|
|
389
|
+
* `assertServedModel` / `checkServedModel` (src/integrity/served-model.ts).
|
|
390
|
+
*
|
|
391
|
+
* Optional so hand-built results (mock/custom transports) still typecheck,
|
|
392
|
+
* but omitting it is not a pass: the identity checks read `undefined` as
|
|
393
|
+
* `unreported` and reject it by default. A transport that knows which model
|
|
394
|
+
* answered should say so.
|
|
395
|
+
*/
|
|
396
|
+
servedModel?: string | null;
|
|
198
397
|
/** Wall-clock duration of the HTTP call (last attempt, if retried). */
|
|
199
398
|
durationMs: number;
|
|
200
399
|
/**
|
|
@@ -307,6 +506,16 @@ interface LlmClientOptions {
|
|
|
307
506
|
};
|
|
308
507
|
/** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
|
|
309
508
|
redactor?: ProviderRedactor;
|
|
509
|
+
/**
|
|
510
|
+
* Reject a response whose echoed model is not the model that was requested.
|
|
511
|
+
* A routing gateway can accept one id and answer from another, which
|
|
512
|
+
* silently invalidates every per-model and per-family claim downstream.
|
|
513
|
+
* `true` uses the strict default (aliases pass, substitutions and
|
|
514
|
+
* unidentified responses throw `ModelSubstitutionError`); pass an options
|
|
515
|
+
* object to relax a specific case. Off by default — turning it on for a
|
|
516
|
+
* measurement run is the point.
|
|
517
|
+
*/
|
|
518
|
+
assertServedModel?: boolean | AssertServedModelOptions;
|
|
310
519
|
}
|
|
311
520
|
/**
|
|
312
521
|
* True when an error is a transient transport/network fault worth retrying,
|
|
@@ -389,10 +598,16 @@ declare function assertLlmRoute(opts: LlmClientOptions, req?: LlmRouteRequiremen
|
|
|
389
598
|
* network, parse). Designed for sweep preflights — fail loud at the
|
|
390
599
|
* boundary before burning a 30-leaf run on a misconfigured router.
|
|
391
600
|
*
|
|
392
|
-
* Sends a tiny `ping` message with `maxTokens=
|
|
393
|
-
* (glm-5.1, deepseek-v4) can burn the entire budget on internal
|
|
394
|
-
* for short prompts, so don't tighten this further
|
|
395
|
-
*
|
|
601
|
+
* Sends a tiny `ping` message with `maxTokens = PROBE_MAX_TOKENS`. Reasoning
|
|
602
|
+
* models (glm-5.1, deepseek-v4) can burn the entire budget on internal
|
|
603
|
+
* reasoning for short prompts, so don't tighten this further — the shared
|
|
604
|
+
* constant keeps this probe and `preflightModels` on one answer. We don't
|
|
605
|
+
* validate content.
|
|
606
|
+
*
|
|
607
|
+
* Reachability and identity are separate answers: `ok` means the route
|
|
608
|
+
* answered, `servedModel` / `substituted` say WHICH model answered. A gateway
|
|
609
|
+
* that serves another provider's model returns `ok: true` with
|
|
610
|
+
* `substituted: true` — inspect both before treating the id as measured.
|
|
396
611
|
*/
|
|
397
612
|
declare function probeLlm(model: string, opts?: LlmClientOptions & {
|
|
398
613
|
timeoutMs?: number;
|
|
@@ -400,6 +615,10 @@ declare function probeLlm(model: string, opts?: LlmClientOptions & {
|
|
|
400
615
|
ok: boolean;
|
|
401
616
|
latencyMs: number;
|
|
402
617
|
error: string | null;
|
|
618
|
+
/** Id echoed by the provider; `null` when it sent none or the probe failed. */
|
|
619
|
+
servedModel: string | null;
|
|
620
|
+
/** True when the echoed id is a different model than `model` (or absent). */
|
|
621
|
+
substituted: boolean;
|
|
403
622
|
}>;
|
|
404
623
|
/**
|
|
405
624
|
* Stateful client — construct once with defaults, call many times.
|
|
@@ -800,5 +1019,5 @@ interface EvalResult {
|
|
|
800
1019
|
artifact?: string;
|
|
801
1020
|
}
|
|
802
1021
|
//#endregion
|
|
803
|
-
export { assertLlmRoute as $, ChatClient as A, SandboxSdkTransportOpts as B, ScenarioFile as C, TurnMetrics as D, Turn as E, CreateChatClientOpts as F, LlmCallResult as G, LlmCallError as H, CustomTransportOpts as I, LlmMessage as J, LlmClient as K, DirectProviderTransportOpts as L, ChatResponse as M, ChatTransport as N, TurnResult as O, CliBridgeTransportOpts as P, LlmUsage as Q, MockTransportOpts as R, Scenario as S, TestResult as T, LlmCallMetadata as U, createChatClient as V, LlmCallRequest as W, LlmRouteAssertionError as X, LlmResponseError as Y, LlmRouteRequirements as Z, PersonaConfig as _,
|
|
804
|
-
//# sourceMappingURL=types-
|
|
1022
|
+
export { assertLlmRoute as $, ChatClient as A, InMemoryRawProviderSinkOptions as At, SandboxSdkTransportOpts as B, ScenarioFile as C, CrossFamilyError as Ct, TurnMetrics as D, FileSystemRawProviderSink as Dt, Turn as E, judgeFamily as Et, CreateChatClientOpts as F, RawProviderSink as Ft, LlmCallResult as G, LlmCallError as H, CustomTransportOpts as I, RawProviderSinkFilter as It, LlmMessage as J, LlmClient as K, DirectProviderTransportOpts as L, defaultProviderRedactor as Lt, ChatResponse as M, ProviderRedactor as Mt, ChatTransport as N, RawProviderDirection as Nt, TurnResult as O, FileSystemRawProviderSinkOptions as Ot, CliBridgeTransportOpts as P, RawProviderEvent as Pt, LlmUsage as Q, MockTransportOpts as R, providerFromBaseUrl as Rt, Scenario as S, AssertCrossFamilyOptions as St, TestResult as T, assertCrossFamily as Tt, LlmCallMetadata as U, createChatClient as V, LlmCallRequest as W, LlmRouteAssertionError as X, LlmResponseError as Y, LlmRouteRequirements as Z, PersonaConfig as _, assertServedModel as _t, CheckResult as a, isTransientLlmError as at, RouteMap as b, normalizeModelId as bt, DriverResult as c, stripFencedJson as ct, FeedbackPattern as d, ModelSubstitutionError as dt, backoffMs as et, JudgeConfig as f, PROBE_MAX_TOKENS as ft, JudgeScore as g, assertCrossFamilyServed as gt, JudgeRubric as h, ServedModelVerdict as ht, BenchmarkRunnerConfig as i, costReceiptFromLlmError as it, ChatRequest as j, NoopRawProviderSink as jt, ChatCallOpts as k, InMemoryRawProviderSink as kt, DriverState as l, AssertCrossFamilyServedOptions as lt, JudgeInput as m, ServedModelCheck as mt, ArtifactResult as n, callLlmJson as nt, CollectedArtifacts as o, maximumChargeForLlmRequest as ot, JudgeFn as p, ServedCrossFamilyError as pt, LlmClientOptions as q, BenchmarkReport as r, costReceiptFromLlm as rt, CompletionCriterion as s, probeLlm as st, ArtifactCheck as t, callLlm as tt, EvalResult as u, AssertServedModelOptions as ut, PersonaRigor as v, assertServedModels as vt, ScenarioResult as w, JudgeFamily as wt, RubricDimension as x, servedModelAcceptable as xt, ProductClientConfig as y, checkServedModel as yt, RouterTransportOpts as z };
|
|
1023
|
+
//# sourceMappingURL=types-D216SgwM.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types-D216SgwM.d.ts","names":[],"sources":["../src/trace/raw-provider-sink.ts","../src/judge-families.ts","../src/integrity/served-model.ts","../src/llm-client.ts","../src/analyst/chat-client.ts","../src/types.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;KA8BY;UAEK;;EAEf;;EAEA;EACA;;;;;;EAMA;EACA;;EAEA;;EAEA;;EAEA;EACA,WAAW;;EAEX;;EAEA;EACA;EACA,iBAAiB;EACjB;EACA,kBAAkB;EAClB;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA,YAAY;EACZ;;UAGe;EACf,OAAO,OAAO,mBAAmB;;EAEjC,MAAM,SAAS,wBAAwB,QAAQ;;EAE/C,UAAU;;KAGA,oBAAoB,OAAO,qBAAqB;;;;;;iBAmB5C,wBAAwB,OAAO,mBAAmB;UA8CjD;EACf,WAAW;;cAGA,mCAAmC;UACtC;UACA;EAER,YAAY,OAAM;EAIZ,OAAO,OAAO,mBAAmB;EAIjC,KAAK,SAAQ,wBAA6B,QAAQ;EAUxD;;cAKW,+BAA+B;EACpC,UAAU;;;;;;;EASV,QAAQ,QAAQ;;UAOP;;EAEf;;EAEA;;EAEA;EACA,WAAW;;cAGA,qCAAqC;UACxC;UACA;UACA;UACA;UACA;UACA;UACA;EAER,YAAY,MAAM;UAOJ;UAON;EAKF,OAAO,OAAO,mBAAmB;EAYjC,KAAK,SAAQ,wBAA6B,QAAQ;;;;;;iBAkC1C,oBAAoB;;;;;;;;;;;;;KC5QxB;;;;;;;iBAkEI,YAAY,kBAAkB;UAc7B;;EAEf;;;;EAIA;;cAGW,yBAAyB;WAGlB,UAAU;WACV;EAHlB,YACE,iBACgB,UAAU,eACV;;;;;;;;;;;;;;;iBAoBJ,kBACd,kBACA,OAAM,2BACL;;;;;;;;;;;;;;;;;;;cCtFU;;KAGD;;;;;;;;;;;UAYK;;EAEf;;EAEA;EACA,iBAAiB;;EAEjB,cAAc;EACd,SAAS;;EAET;;;;;;;;;;;;;;;;;iBAkBc,iBAAiB;;;;;;;;;;;iBA0BjB,iBACd,mBACA,oCACC;cA4CU,+BAA+B;WAGxB,QAAQ,cAAc;EAFxC,YACE,iBACgB,QAAQ,cAAc;;UAOzB;;;;;;EAMf;;;;;EAKA;;EAEA;;;;;;;iBAQc,sBACd,OAAO,kBACP,OAAM;;;;;;iBA6BQ,kBACd,mBACA,mCACA,OAAM,2BACL;;;;;iBAea,mBACd,OAAO;EAAgB;EAAmB;IAC1C,OAAM,2BACL;UAec,uCAAuC;;EAEtD;;EAEA;;cAGW,+BAA+B;WAGxB,UAAU;WACV,QAAQ,cAAc;EAHxC,YACE,iBACgB,UAAU,eACV,QAAQ,cAAc;;;;;;;;;;;iBAgB1B,wBACd,OAAO;EAAgB;EAAmB;IAC1C,OAAM,iCACL;;;UCjPc;EACf;;;;;EAKA,kBAEI;IACM;IAAc;;IACd;IAAmB;MAAa;MAAa;;;;KAI7C;UAEK;EACf;EACA,UAAU;;EAEV;;EAEA;IAAe;IAAc,QAAQ;;EACrC;EACA;;EAEA,WAAW;;EAEX;;;;;;iBAOc,2BACd,SAAS,KAAK,iFACd,UAAS,mBACR;UAgCc;EACf;EACA;EACA;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf;EACA,OAAO;;;;;EAKP;;;;;;;;;EASA;;;;;;;;;;;;EAYA;;EAEA;;;;;;;;;EASA;;;;;;;EAOA;;EAEA,KAAK;;KAGK,kBAAkB,KAAK;;iBAGnB,mBACd,QAAQ,eACR,qBAAqB,qBACpB;;iBA2Ba,wBACd,OAAO,OACP,qBAAqB,qBACpB;cAMU,qBAAqB;WAGd;WACA;WACA;EAJlB,YACE,iBACgB,gBACA,cACA;;;;;cASP,yBAAyB;WAGlB,QAAQ;EAF1B,YACE,iBACgB,QAAQ,eACxB;IAAY;;;UAMC;;EAEf;;EAEA;EACA;;EAEA;IAAe;IAAc;;;EAE7B;;EAEA;;;;;;;EAOA,SAAS;;;;;;;;EAQT;;EAEA;;EAEA,qBAAqB;;;;;;;EAOrB;;;;;;EAMA;;EAEA,WAAW;;EAEX,eAAe;;;;;;;;EAQf,UAAU;;;;;EAKV;;EAEA;IAAiB;IAAgB;;;EAEjC,WAAW;;;;;;;;;;EAUX,8BAA8B;;;;;;;;;;;;;iBAsEhB,oBAAoB;;iBAgCpB,UAAU;;;;;;iBA6GV,gBAAgB;;;;;;iBA6EV,QACpB,KAAK,gBACL,OAAM,mBACL,QAAQ;;;;;;;iBAoVW,YAAY,aAChC,KAAK,gBACL,OAAM,mBACL;EAAU,OAAO;EAAG,QAAQ;;KA4DnB;cAOC,+BAA+B;WAGxB,QAAQ;WACR;EAHlB,YACE,iBACgB,QAAQ,yBACR;;UAMH;;;;;;;EAOf;;;;;EAKA,kBAAkB,eAAe;;EAEjC,kBAAkB,eAAe;;EAEjC;;;;;EAKA;;;;;;;;;;;iBAYc,eAAe,MAAM,kBAAkB,MAAK;;;;;;;;;;;;;;;;;;iBA6EtC,SACpB,eACA,OAAM;EAAqB;IAC1B;EACD;EACA;EACA;;EAEA;;EAEA;;;;;;;cAoCW;WACF;mBACQ;EAEjB,YAAY,OAAM;EAKlB,KAAK,KAAK,gBAAgB,MAAM,mBAAmB,QAAQ;EAK3D,SAAS,aACP,KAAK,gBACL,MAAM,mBACL;IAAU,OAAO;IAAG,QAAQ;;;;;;;;UCjqChB;;WAEN,WAAW;;WAEX;;WAEA;;EAGT,KAAK,KAAK,aAAa,OAAO,eAAe,QAAQ;;KAG3C;UAQK,oBAAoB,KAAK;;EAExC;;KAGU,eAAe;UAEV;;EAEf,SAAS;;EAET;;EAEA;;EAEA;;KAKU,uBACR,sBACA,yBACA,8BACA,0BACA,sBACA;UAEM;EACR;;EAEA;;UAGe,4BAA4B;EAC3C;EACA;EACA;;UAGe,+BAA+B;EAC9C;EACA;EACA;;UAGe,oCAAoC;EACnD;EACA;EACA;;;;;;UAOe,gCAAgC;EAC/C;EACA,OAAO,KAAK,aAAa,OAAO,iBAAiB,QAAQ;;;UAI1C,4BAA4B;EAC3C;EACA,OAAO,KAAK,aAAa,OAAO,iBAAiB,QAAQ;;;;;;UAO1C,0BAA0B;EACzC;EACA,UAAU,KAAK,aAAa,OAAO,iBAAiB,QAAQ;;;;;;iBAO9C,iBAAiB,MAAM,uBAAuB;;;UCjH7C;EACf;EACA;EACA;EACA;EACA;EACA,OAAO;EACP,gBAAgB;EAChB;;UAGe;EACf;EACA;EACA;EACA;;UAKe;EACf;EAQA;EACA;EACA;EACA;;UAKe;EACf;EACA;EACA,QAAQ;;UAGO;EACf;EACA;EACA,YAAY;;UAGG;EACf;EACA;EACA;EACA;EACA;;UAKe;EACf;EACA;EACA,OAAO;EACP,iBAAiB;EACjB,aAAa;EACb;EACA;EACA;EACA,WAAW;;EAEX,OAAO;;UAGQ;EACf;EACA;EACA;EACA;EACA;IAAmB;IAAc;;EACjC;EACA;;UAGe;EACf,OAAO;EACP;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;IAAc;IAAc;;EAC5B;IAAmB;IAAc,QAAQ;;EACzC;IAAc;IAAkB;;EAChC;;UAKe;EACf;EACA;EACA;EACA;EACA,SAAS;EACT,OAAO;EACP;IACE;IACA,WAAW;MAAiB;MAAa;MAAgB;;IACzD,aAAa;MAAiB;MAAa;;IAC3C;MAAW;MAAkB;MAAe;;IAC5C;MAAa;MAAkB;MAAe;;;;UAMjC;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;GACC;;UAGc;EACf;EACA,QAAQ;;EAER;;UAKe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;IACE;MACE;MACA;MACA;;;EAGJ,OAAO;EACP,gBAAgB;;UAKD;EACf;EACA,QAAQ,OAAO;EACf,YAAY,OAAO;;UAGJ;EACf;EACA;;;;;;;;;;KAWU;UAEK;EACf;EACA;EACA;EACA,oBAAoB;EACpB,mBAAmB;EACnB;EACA;;EAEA,QAAQ;;;;;;;EAOR;;;;;;;EAOA;;;;;;EAMA;;UAGe;EACf;EACA;EACA;IAAa;IAAiB;IAAkB;;EAChD;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;IAAa;IAAiB;IAAkB;;EAChD;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;;EAEA;;EAEA;;;;;;EAMA;EACA;EACA,SAAS;EACT,YAAY;EACZ;EACA;EACA;;UAKe;EACf,WAAW;EACX,QAAQ;EACR;EACA;EACA;EACA;EACA;EACA;;EAEA,aAAa;;UAGE;EACf,UAAU;EACV,OAAO;EACP,WAAW;;EAEX,aAAa;EACb;EACA,WAAW;EACX,SAAS;;KAGC,WAAW,MAAM,YAAY,OAAO,eAAe,QAAQ;UAItD;EACf;EACA;EACA;EACA;EACA,QAAQ;;UAGO;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { h as RolloutLine } from "./schema-
|
|
1
|
+
import { h as RolloutLine } from "./schema-Cef2cFmb2.js";
|
|
2
2
|
//#region src/supervisor-run/types.d.ts
|
|
3
3
|
/** A metric that could not be computed, with the reason its artifact was missing. */
|
|
4
4
|
interface Unavailable {
|
|
@@ -369,4 +369,4 @@ interface SupervisorRunTreeGap {
|
|
|
369
369
|
}
|
|
370
370
|
//#endregion
|
|
371
371
|
export { Unavailable as C, showMeasured as D, isUnavailable as E, unavailable as O, SupervisorRunTreeGapCode as S, WorkerLogSource as T, SupervisorRunReport as _, OrchestrationMetrics as a, SupervisorRunTree as b, PerWorkerRow as c, SUPERVISOR_RUN_ROLLUP_SCHEMA as d, SUPERVISOR_RUN_SCHEMA as f, SupervisorRunReader as g, SupervisorRunNodeRole as h, NO_SOURCE_LIMITS as i, RoleSpend as l, SteerBreakdown as m, EconomicsMetrics as n, OutcomeMetrics as o, SourceLimits as p, Measured as r, PatchStats as s, DecisionMetrics as t, RollupCellRow as u, SupervisorRunRollup as v, WallDistribution as w, SupervisorRunTreeGap as x, SupervisorRunSources as y };
|
|
372
|
-
//# sourceMappingURL=types-
|
|
372
|
+
//# sourceMappingURL=types-D4mog56g.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types-
|
|
1
|
+
{"version":3,"file":"types-D4mog56g.d.ts","names":[],"sources":["../src/supervisor-run/types.ts"],"mappings":";;;UAiCiB;WACN;;;KAIC,SAAS,KAAK,IAAI;iBAEd,YAAY,iBAAiB;iBAI7B,cAAc,aAAa,KAAK;;iBAKhC,aAAa,GAAG;;KAWpB;;;;;UAMK;;;;;WAKN;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;WACA;WACA;WACA;;;;;;;;;;;;;UAcM;;WAEN;;WAEA;;WAEA;;WAEA;;WAEA;;;cAIE,kBAAkB;;;;;;;;;;UAiBd;;WAEN;WACA;;WAEA;;WAEA;;;;;;WAMA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA,kBAAkB;;WAElB;;WAEA;;;;;;;WAOA;;WAEA;;WAEA;;WAEA;;;;;WAKA;IACP;IACA;IACA;IACA;;IAEA;IACA;;WAEO;;WAEA,QAAQ;;;;;;WAMR;;;;;WAKA;;;;;;UAOM;;WAEN;EACT,QAAQ,QAAQ;;cAOL;cACA;UAEI;;WAEN;WACA;;WAEA;;WAEA;;UAGM;WACN,gBAAgB;WAChB,gBAAgB;WAChB,kBAAkB;;WAElB,QAAQ;WACR,iBAAiB;WACjB,gBAAgB,kBAAkB;;WAElC,kBAAkB;;;;;;WAMlB,OAAO;WACP,WAAW;WACX,gBAAgB;;WAEhB,UAAU;;WAEV,gBAAgB;;WAEhB,iBAAiB;WACjB,oBAAoB;WACpB,kBAAkB;;WAElB,QAAQ;WACR,SAAS;;WAET,mBAAmB;;UAGb;WACN,iBAAiB,SAAS;WAC1B,iBAAiB,SAAS;;WAE1B,UAAU;;WAEV,UAAU;;WAEV,WAAW;;WAEX,oBAAoB;;WAEpB,wBAAwB;;WAExB,eAAe;WACf,qBAAqB;;UAGf;WACN,UAAU;WACV,WAAW;;;;;;WAMX,WAAW;WACX,YAAY;WACZ,KAAK;WACL;;UAGM;;WAEN;WACA;;WAEA,MAAM;;WAEN;;WAEA;;WAEA;;WAEA;;WAEA;WACA;;WAEA;WACA;WACA;WACA;WACA;;WAEA;;UAGM;WACN;WACA;WACA;WACA;WACA;WACA;;UAGM;;WAEN,OAAO;;;;;;;;WAQP,kBAAkB;;WAElB,SAAS;WACT,UAAU;;;;;;WAMV;WACA,yBAAyB;WACzB,0BAA0B,SAAS;WACnC,WAAW,kBAAkB;;UAGvB;WACN;WACA;WACA;WACA;;UAGM;WACN,WAAW;WACX,YAAY;WACZ,WAAW;WACX,eAAe;WACf,YAAY;WACZ,aAAa;WACb,YAAY;WACZ,YAAY;WACZ,UAAU;WACV,OAAO,SAAS;;WAEhB;;UAGM;WACN,eAAe;;WAEf;WACA;WACA;WACA,cAAc;WACd,yBAAyB;WACzB;WACA,eAAe;WACf,UAAU;WACV,WAAW;WACX,SAAS;;WAET;;WAEA;;UAGM;WACN;WACA;WACA,QAAQ;WACR,OAAO;WACP,aAAa;WACb,SAAS;WACT,UAAU;WACV,KAAK;;UAGC;WACN,eAAe;WACf;WACA,aAAa;WACb,iBAAiB;WACjB;WACA,WAAW;WACX,mBAAmB;WACnB,iBAAiB;WACjB,aAAa;WACb,qBAAqB;WACrB,eAAe;WACf,UAAU;WACV,eAAe;WACf,kBAAkB;;;;;;;;UASZ;WACN;WACA,gBAAgB;;WAEhB,eAAe;;;KAId;UASK;WACN,MAAM;WACN;WACA;WACA"}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { c as CostLedgerHandle } from "./cost-ledger-
|
|
2
|
-
import { a as RunRecord, n as RunCostProvenance, u as RunTokenUsage } from "./run-record-
|
|
3
|
-
import { A as ChatClient, m as JudgeInput } from "./types-
|
|
1
|
+
import { c as CostLedgerHandle } from "./cost-ledger-Bv_e8XHY.js";
|
|
2
|
+
import { a as RunRecord, n as RunCostProvenance, u as RunTokenUsage } from "./run-record-DdSa93_W.js";
|
|
3
|
+
import { A as ChatClient, m as JudgeInput } from "./types-D216SgwM.js";
|
|
4
4
|
import { RE2JS } from "re2js";
|
|
5
5
|
//#region src/trace-analyst/types.d.ts
|
|
6
6
|
/**
|
|
@@ -400,6 +400,55 @@ interface AnalystContext {
|
|
|
400
400
|
log?: (msg: string, fields?: Record<string, unknown>) => void;
|
|
401
401
|
/** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
|
|
402
402
|
signal?: AbortSignal;
|
|
403
|
+
/**
|
|
404
|
+
* Optional live-execution port. A runtime that owns a sandbox or checkout
|
|
405
|
+
* fills it so an analyst can execute a bounded probe against the run's
|
|
406
|
+
* produced state instead of reasoning about it from the trace alone. This
|
|
407
|
+
* package defines only the port: no field here reaches for an agent loop,
|
|
408
|
+
* and an absent probe means the analyst works from recorded evidence.
|
|
409
|
+
*/
|
|
410
|
+
probe?: ExecutionProbe;
|
|
411
|
+
}
|
|
412
|
+
/** One bounded command an analyst asks the probe to run. */
|
|
413
|
+
interface ExecutionProbeRequest {
|
|
414
|
+
command: string;
|
|
415
|
+
/** Working directory inside the probed environment. */
|
|
416
|
+
cwd?: string;
|
|
417
|
+
/** Hard wall-clock deadline for this one execution. */
|
|
418
|
+
timeoutMs: number;
|
|
419
|
+
/** Bytes of combined output retained; the prober truncates beyond it. */
|
|
420
|
+
maxOutputBytes?: number;
|
|
421
|
+
signal?: AbortSignal;
|
|
422
|
+
}
|
|
423
|
+
/**
|
|
424
|
+
* Typed outcome of one probe execution. `succeeded: false` is a PROBE failure
|
|
425
|
+
* (the environment could not run the command); a command that ran and exited
|
|
426
|
+
* non-zero is a successful observation with a non-zero `exitCode`.
|
|
427
|
+
*/
|
|
428
|
+
type ExecutionProbeOutcome = {
|
|
429
|
+
succeeded: true;
|
|
430
|
+
exitCode: number;
|
|
431
|
+
stdout: string;
|
|
432
|
+
stderr: string;
|
|
433
|
+
durationMs: number;
|
|
434
|
+
/** True when output was cut at `maxOutputBytes`. */
|
|
435
|
+
truncated: boolean;
|
|
436
|
+
} | {
|
|
437
|
+
succeeded: false;
|
|
438
|
+
error: {
|
|
439
|
+
class: string;
|
|
440
|
+
message: string;
|
|
441
|
+
};
|
|
442
|
+
};
|
|
443
|
+
/**
|
|
444
|
+
* The seam a runtime fills to let analysts observe produced state live.
|
|
445
|
+
* Implementations own sandboxing, credentials, and cleanup; analysts only
|
|
446
|
+
* submit bounded requests and read typed outcomes.
|
|
447
|
+
*/
|
|
448
|
+
interface ExecutionProbe {
|
|
449
|
+
/** One plain sentence naming what is being probed (e.g. a sandbox id). */
|
|
450
|
+
readonly description: string;
|
|
451
|
+
execute(request: ExecutionProbeRequest): Promise<ExecutionProbeOutcome>;
|
|
403
452
|
}
|
|
404
453
|
/**
|
|
405
454
|
* The minimal contract. Concrete analysts can refine `TInput` so
|
|
@@ -544,5 +593,5 @@ type AnalystRunEvent = {
|
|
|
544
593
|
result: AnalystRunResult;
|
|
545
594
|
};
|
|
546
595
|
//#endregion
|
|
547
|
-
export {
|
|
548
|
-
//# sourceMappingURL=types-
|
|
596
|
+
export { SearchSpanResult as A, ViewSpansResult as B, TRACE_ANALYSIS_LIMITS as C, DatasetOverview as D, DEFAULT_TRACE_ANALYST_BUDGETS as E, TraceAnalystFilters as F, ViewTraceResult as H, TraceAnalystSpan as I, TraceAnalystSpanKind as L, SpanMatchRecord as M, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX as N, ErrorCluster as O, TraceAnalystByteBudgets as P, TraceAnalystSpanStatus as R, BoundedTraceAnalysisStoreOptions as S, TraceAnalysisStoreContext as T, ViewTraceOversized as V, ProposalFinding as _, AnalystInputKind as a, makeFinding as b, AnalystRunInputs as c, AnalystSeverity as d, AnalystUsageReceipt as f, ExecutionProbeRequest as g, ExecutionProbeOutcome as h, AnalystFinding as i, SearchTraceResult as j, QueryTracesPage as k, AnalystRunResult as l, ExecutionProbe as m, AnalystContext as n, AnalystRequirements as o, EvidenceRef as p, AnalystCost as r, AnalystRunEvent as s, Analyst as t, AnalystRunSummary as u, ProposalFindingOrigin as v, TraceAnalysisStore as w, makeProposalFinding as x, computeFindingId as y, TraceAnalystTraceSummary as z };
|
|
597
|
+
//# sourceMappingURL=types-DF_Udrp-.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types-
|
|
1
|
+
{"version":3,"file":"types-DF_Udrp-.d.ts","names":[],"sources":["../src/trace-analyst/types.ts","../src/trace-analyst/store-contract.ts","../src/analyst/types.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;KAgBY;KAUA;;;;UAKK;EACf;EACA;EACA;EACA;EACA,MAAM;EACN;EACA;EACA;EACA,QAAQ;EACR;EACA;EACA;EACA;EACA;;;EAGA,YAAY;;UAGG;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;;;;;UAQe;;EAEf;;EAEA;;EAEA;;EAEA;EACA;EACA;;EAEA;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;;EAEA;EACA;IAAU;IAAqB;;;;;;EAK/B,gBAAgB;EAChB;IAAc;IAAkB;;;UAGjB;EACf,QAAQ;EACR;EACA;;;;;UAMe;EACf;EACA,QAAQ;EACR,YAAY;;UAGG;EACf;;EAEA,gBAAgB;;EAEhB;EACA;;UAGe;EACf;EACA,OAAO;;EAEP;;EAEA;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA;EACA,WAAW;;;EAGX;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA,MAAM;EACN;;UAGe;EACf;EACA;EACA,MAAM;EACN;;;UAIe;;;EAGf;;;EAGA;;;EAGA;;;EAGA;;cAGW,+BAA+B;;;cAS/B;;;cC7MA;WACX;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;UAGe;EACf,SAAS;;;;;;;;;UAUM;EACf,SAAS,kBAAkB,UAAU,4BAA4B;EAEjE,SACE;IAAS;IAAkB;KAC3B,UAAU,4BACT;EAEH,YACE,UAAU,qBACV,UAAU,4BACT,QAAQ;EAEX,YACE;IAAS,UAAU;IAAqB;IAAe;KACvD,UAAU,4BACT,QAAQ;EAEX,YAAY,UAAU,qBAAqB,UAAU,4BAA4B;EAEjF,UACE;IACE;IACA;KAEF,UAAU,4BACT,QAAQ;EAEX,UACE;IACE;IACA;IACA;KAEF,UAAU,4BACT,QAAQ;EAEX,YACE;IACE;IACA;IACA;KAEF,UAAU,4BACT,QAAQ;EAEX,WACE;IACE;IACA;IACA;IACA;KAEF,UAAU,4BACT,QAAQ;;UAGI;EACf,UAAU,QAAQ;;;;;;;;UC/DH;EACf;;;;;;;EAOA;EACA;EACA;EACA,UAAU;;;;;;;EAOV;EACA;EACA;EACA,eAAe;EACf;EACA;;EAEA;;;;;;EAMA;;;;EAIA;;EAEA,WAAW;;KAGD;;KAGA;;KAGA,kBAAkB;WACnB,iBAAiB;;UAGX;;;;;;;EAOf;EACA;EACA;;;;;;;;KAWU;UAOK;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf;;EAEA;;;;;;UAOe;EACf,aAAa;EACb;EACA,YAAY;EACZ,aAAa;;EAEb,SAAS;;UAGM;EACf;;EAEA;;EAEA;;EAEA;;EAEA,aAAa;;EAEb;;;;;;;EAOA,OAAO;;;;;;;;;;EAUP,gBAAgB,cAAc;;;;;;;EAO9B,mBAAmB,cAAc;;;;;EAKjC,eAAe,SAAS;;EAExB,OAAO;;EAEP,OAAO,aAAa,SAAS;;EAE7B,SAAS;;;;;;;;EAQT,QAAQ;;;UAMO;EACf;;EAEA;;EAEA;;EAEA;EACA,SAAS;;;;;;;KAQC;EAEN;EACA;EACA;EACA;EACA;;EAEA;;EAEA;EAAkB;IAAS;IAAe;;;;;;;;UAO/B;;WAEN;EACT,QAAQ,SAAS,wBAAwB,QAAQ;;;;;;;;UASlC,QAAQ;;WAEd;;WAEA;WACA,WAAW;WACX,MAAM;WACN,WAAW;;WAEX;EACT,QAAQ,OAAO,QAAQ,KAAK,iBAAiB,QAAQ;;;UAItC;;EAEf;;EAEA,QAAQ;;EAER,MAAM;;EAEN;;;;;;;;EAQA;IAAkB;IAAsB;;;;;;;;EAOxC;;;;;;;;;;iBAac,iBAAiB;EAC/B;EACA;EACA;EACA;;EAEA;;;;;;iBAyBc,YACd,MAAM,KAAK;EACT;EACA;IAED;;iBAiBa,oBACd,MAAM,KAAK;EACT;EACA;IAED;UAOc;EACf;EACA;;EAEA;EACA;EACA;;EAEA,OAAO;;EAEP;IAAU;IAAe;;;UAGV;EACf;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa;;EAEb;;;;;EAKA,wBAAwB;;;;;;;;;;;;;;;KAkBd;EAEN;EACA;EACA;EACA;;EAEA,aAAa;;EAGb;EACA,SAAS;;EAGT;EACA;EACA;;EAGA;;EAEA,SAAS;EACT,UAAU,cAAc;;EAGxB;EACA,QAAQ"}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { S as PaidCallResult, T as RunPaidCallInput, a as CostChannel, c as CostLedgerHandle, f as CostLedgerSummary, p as CostProvenance } from "./cost-ledger-
|
|
2
|
-
import { u as RunTokenUsage } from "./run-record-
|
|
3
|
-
import { U as LlmCallMetadata } from "./types-
|
|
4
|
-
import {
|
|
1
|
+
import { S as PaidCallResult, T as RunPaidCallInput, a as CostChannel, c as CostLedgerHandle, f as CostLedgerSummary, p as CostProvenance } from "./cost-ledger-Bv_e8XHY.js";
|
|
2
|
+
import { u as RunTokenUsage } from "./run-record-DdSa93_W.js";
|
|
3
|
+
import { U as LlmCallMetadata } from "./types-D216SgwM.js";
|
|
4
|
+
import { _ as ProposalFinding } from "./types-DF_Udrp-.js";
|
|
5
5
|
//#region src/campaign/types.d.ts
|
|
6
6
|
/** Stable identifier + kind tag for any scenario. Consumers
|
|
7
7
|
* extend with their per-domain payload (persona, task, requirement, ...). */
|
|
@@ -624,4 +624,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
|
|
|
624
624
|
}
|
|
625
625
|
//#endregion
|
|
626
626
|
export { LabeledScenarioWrite as A, ScoredSurfaceOutcome as B, JudgeDimension as C, LabeledScenarioSampleArgs as D, LabeledScenarioRecord as E, ProposeContext as F, labelTrustRank as G, SurfaceProposer as H, ProposedCandidate as I, RedactionStatus as L, OptimizerConfig as M, ParetoParent as N, LabeledScenarioSource as O, ProposalTrackContext as P, Scenario as R, JudgeConfig as S, LabelTrust as T, TraceSpan as U, SessionScript as V, isProposedCandidate as W, GateDecision as _, CampaignResult as a, GenerationRecord as b, CampaignTraceWriter as c, DispatchContext as d, DispatchFn as f, GateContribution as g, GateContext as h, CampaignCostMeter as i, MutableSurface as j, LabeledScenarioStore as k, CodeSurface as l, GateCheckStatus as m, CampaignArtifactWriter as n, CampaignScenarioIdentity as o, Gate as p, CampaignCellResult as r, CampaignTokenUsage as s, CampaignAggregates as t, ComponentSurface as u, GateResult as v, JudgeScore as w, JudgeAggregate as x, GenerationCandidate as y, ScenarioAggregate as z };
|
|
627
|
-
//# sourceMappingURL=types-
|
|
627
|
+
//# sourceMappingURL=types-DYuNHo9R.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types-
|
|
1
|
+
{"version":3,"file":"types-DYuNHo9R.d.ts","names":[],"sources":["../src/campaign/types.ts"],"mappings":";;;;;;;UAgCiB;EACf;EACA;EACA;;;;;EAKA;;;UAIe,iCAAiC,KAAK;EACrD;;;;;UAMe;EACf;EACA;EACA;EACA;EACA,QAAQ;EACR,OAAO;EACP,WAAW;EACX,MAAM;;EAEN;;EAEA;;;;;;;;EAQA;;;;KAKU,WAAW,kBAAkB,UAAU,cACjD,UAAU,WACV,KAAK,oBACF,QAAQ;;;;UAOI,cAAc,WAAW;EACxC;EACA;EACA;;EAEA;;;EAGA,sBAAsB,UAAU,WAAW,sBAAsB,UAAU,cAAc;;UAK1E;;EAEf;;EAEA;;;;;;;;;UAUe,YAAY,WAAW,kBAAkB,WAAW;EACnE;EACA,YAAY;;;;EAIZ;;;EAGA,MAAM;IACJ,UAAU;IACV,UAAU;IACV,QAAQ;;IAER,aAAa;IACb;IACA,WAAW;MACT,aAAa,QAAQ;EACzB,aAAa,UAAU;;;;;;;;;;;UAYR;EACf,YAAY;EACZ;EACA;;EAEA,UAAU;;;;EAIV;;;EAGA;;EAEA;;EAEA,WAAW,eAAe;;;;;;;UAUX;WACN;;;WAGA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;;WAGA;aACE;aACA;aACA;;;WAGF;;;UAIM;WACN;WACA,YAAY,SAAS;;;;;;;;;;;KAYpB,0BAA0B,mBAAmB;;;;;;;UAQxC;EACf,SAAS;;EAET;;;;EAIA;;;;iBAKc,oBACd,OAAO,iBAAiB,oBACvB,SAAS;;;;;;;;;UAkBK;EACf,SAAS;EACT;;;EAGA,YAAY;;;EAGZ;;EAEA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EACA;;;;;UAMe;;;EAGf;;EAEA;EACA;EACA;EACA,YAAY;;;;;EAKZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;EAC1E;IACE;IACA;;;;;UAMa,eAAe,YAAY;WACjC,gBAAgB;WAChB,SAAS,cAAc;WACvB,UAAU,cAAc;;WAExB;WACA;WACA,QAAQ;;WAER,QAAQ;;;WAGR,kBAAkB;;;WAGlB,mBAAmB;;;;WAInB;;;;;;;WAOA,gBAAgB,cAAc;;WAE9B,aAAa;WACb;;;;;;;;;;;;;;UAeM,gBAAgB,YAAY;EAC3C;;;;;EAKA,QAAQ,KAAK,eAAe,aAAa,QAAQ,MAAM,iBAAiB;;;EAGxE,QAAQ;IAAQ,SAAS,cAAc;;IAAwB;IAAe;;;UAG/D;EACf;EACA;EACA,mBAAmB,qBAAqB;;UAGzB,wBAAwB;EACvC,UAAU;;;KAMA;;KAGA;UAEK;EACf;EACA,QAAQ;EACR;;UAGe,YAAY,WAAW,kBAAkB;EACxD,oBAAoB,YAAY;EAChC,oBAAoB,YAAY;;EAEhC,aAAa,YAAY,eAAe;;;;;EAKxC,sBAAsB,YAAY,eAAe;;;;;;;EAOjD,yBAAyB,YAAY,eAAe;;;EAGpD,uBAAuB,YAAY;EACnC,WAAW;EACX;IAAQ;IAAmB;;;EAE3B,aAAa;EACb;EACA,QAAQ;;UAGO;EACf,UAAU;EACV;EACA,mBAAmB;EACnB;;;UAIe,KAAK,qBAAqB,kBAAkB,WAAW;EACtE;EACA,OAAO,KAAK,YAAY,WAAW,aAAa,QAAQ;;;;UAOzC;EACf,KAAK,cAAc,aAAa,0BAA0B;EAC1D,SAAS;;UAGM;EACf,IAAI,aAAa;EACjB,aAAa,aAAa;;;;UAKX;EACf,MAAM,cAAc,kBAAkB,aAAa;EACnD,UAAU,cAAc,iBAAiB;;;;;;KAO/B,qBAAqB;;;;;UAMhB;;EAEf,YAAY,GACV,OAAO,KAAK,iBAAiB;IAC3B,UAAU;MAEX,QAAQ,eAAe;;;;;KAQhB;KAOA;;;;;;;;;;;;;;;KAgBA;;iBASI,eAAe,OAAO;;;;UAOrB,qBAAqB,kBAAkB,WAAW,UAAU;EAC3E,UAAU;EACV,UAAU;EACV,aAAa,eAAe;EAC5B,QAAQ;EACR;EACA;EACA,iBAAiB;;;;;EAKjB,aAAa;;EAEb;;UAGe,sBAAsB,kBAAkB,WAAW,UAAU,6BACpE,qBAAqB,WAAW;;EAExC;;;EAGA;;UAGe;EACf;;EAEA;;;;EAIA;EACA;IACE;IACA,SAAS,wBAAwB;IACjC;IACA;;;;;IAKA,WAAW;;;UAIE;EACf,QAAQ,OAAO,uBAAuB;EACtC,OAAO,MAAM,4BAA4B,QAAQ;EACjD,QAAQ;IACN;IACA;IACA,UAAU;;;IAGV,SAAS,OAAO;;;UAMH,mBAAmB;;;EAGlC;EACA;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa,eAAe;;;EAG5B;;EAEA,gBAAgB;;EAEhB;;;EAGA,YAAY;;;EAGZ;;;EAGA;EACA;EACA;EACA;;EAEA;;EAEA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA,YAAY;EACZ;;;;;;UAOe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;;EAIA;;;;EAIA;IACE;IACA;IACA,iBAAiB;MAAQ;MAAgB;;;;;EAI3C,YAAY;;;;;;;;;;EAUZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;;;EAG1E;;;;EAIA;;UAGe;EACf,SAAS,eAAe;EACxB,YAAY,eAAe;;EAE3B,MAAM;;EAEN;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe,eAAe,qBAAqB,kBAAkB,WAAW;;EAEhF;;EAEA;EACA;;EAEA;EACA;EACA;EACA;EACA,OAAO,MAAM,mBAAmB;EAChC,YAAY;EACZ;IACE,aAAa;IACb;;EAEF,OAAO;EACP;EACA;EACA,iBAAiB;;;EAGjB,WAAW,MAAM,2BAA2B,KAAK"}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"usage-receipt-EVI8B8Xu.js","names":[],"sources":["../src/analyst/types.ts","../src/analyst/usage-receipt.ts"],"sourcesContent":["/**\n * Analyst contract — the missing orchestration layer over agent-eval's\n * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,\n * SemanticConceptJudge, JudgeFn, ...).\n *\n * Each existing primitive returns its own output shape. The Analyst\n * contract is the single envelope every primitive lifts into, so a\n * registry can run N analysts against a run and a single renderer can\n * compose findings without knowing which analyzer produced them.\n *\n * The contract is intentionally domain-agnostic: nothing here knows\n * about code, voice, RAG, or any particular agent stack. Analysts\n * declare what INPUT KIND they need (a trace store, an artifact dir,\n * a RunRecord, a JudgeInput, or `custom`), and the registry routes\n * the matching input from `AnalystRunInputs`.\n */\n\nimport { createHash } from 'node:crypto'\nimport type { CostLedgerHandle } from '../cost-ledger'\nimport type { RunCostProvenance, RunRecord, RunTokenUsage } from '../run-record'\nimport type { TraceAnalysisStore } from '../trace-analyst/store'\nimport type { JudgeInput } from '../types'\nimport type { ChatClient } from './chat-client'\n\n/**\n * Unified envelope every analyst emits. Schema-versioned so renderers\n * and time-series diffs survive future field additions.\n */\nexport interface AnalystFinding {\n schema_version: '1.0.0'\n /**\n * Stable hash over identity-defining fields (analyst_id + canonical\n * claim + area + optional subject). Two findings from two runs that\n * \"are the same finding\" share this id — that's what `diffFindings`\n * uses to compute appeared/disappeared sets across runs.\n */\n finding_id: string\n analyst_id: string\n produced_at: string\n severity: AnalystSeverity\n /**\n * Coarse classification. Renderers group by this. Free-form so\n * domain-specific analysts can introduce categories without a\n * schema change ('agent-reasoning', 'verification', 'cost',\n * 'tool-use', 'safety', 'latency', 'data-quality', ...).\n */\n area: string\n claim: string\n rationale?: string\n evidence_refs: EvidenceRef[]\n recommended_action?: string\n validation_plan?: string\n /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */\n confidence: number\n /**\n * Optional subject the finding is about — leaf id, agent id, request\n * id. Included in finding_id when present so per-subject findings\n * diff cleanly across runs.\n */\n subject?: string\n /** True when this finding was lifted from a judge result rather than observed\n * directly in a trace or artifact. Descriptive only: proposal access is\n * controlled by `ProposalFinding.proposal_origin`. */\n derived_from_judge?: boolean\n /** Analyst-private extras; renderers ignore unless they know the analyst. */\n metadata?: Record<string, unknown>\n}\n\nexport type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info'\n\n/** Data sources that candidate generation may intentionally learn from. */\nexport type ProposalFindingOrigin = 'search' | 'production'\n\n/** A finding explicitly admitted as candidate-generation input. */\nexport type ProposalFinding = AnalystFinding & {\n readonly proposal_origin: ProposalFindingOrigin\n}\n\nexport interface EvidenceRef {\n /**\n * Where the evidence lives. `span` and `event` refer to OTLP trace\n * elements; `artifact` to a file inside the run's artifact tree;\n * `finding` to another AnalystFinding (cross-analyst chaining);\n * `metric` to a named scalar reading the renderer knows how to read.\n */\n kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric'\n uri: string\n excerpt?: string\n}\n\n// ── Analyst contract ─────────────────────────────────────────────────\n\n/**\n * The discriminator the registry uses to pass the right input.\n * `custom` is the escape hatch — analysts that need something else\n * (e.g. an embedding cache, a partner SDK handle) read it from\n * `AnalystRunInputs.custom[<analyst id>]`.\n */\nexport type AnalystInputKind =\n | 'trace-store'\n | 'artifact-dir'\n | 'run-record'\n | 'judge-input'\n | 'custom'\n\nexport interface AnalystCost {\n /** `deterministic` analysts MUST NOT call the LLM. */\n kind: 'deterministic' | 'llm'\n /** Optional declared upper bound; the registry can enforce a budget. */\n est_usd_per_run?: number\n /** Models the analyst expects to use (informational). */\n models?: string[]\n /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */\n settlement_timeout_ms?: number\n}\n\nexport interface AnalystRequirements {\n /** Min number of shots / samples the analyst needs to produce signal. */\n min_shots?: number\n /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */\n capabilities?: string[]\n}\n\n/**\n * What's passed to every analyst call. The registry resolves which\n * field the analyst's `inputKind` selects and asserts it's present.\n */\nexport interface AnalystRunInputs {\n traceStore?: TraceAnalysisStore\n artifactDir?: string\n runRecord?: RunRecord\n judgeInput?: JudgeInput\n /** Keyed by analyst id; populated by callers that registered custom analysts. */\n custom?: Record<string, unknown>\n}\n\nexport interface AnalystContext {\n runId: string\n /** Stable correlation id so logs from a single registry.run() share a tag. */\n correlationId: string\n /** Enforced wall-clock deadline (epoch ms). */\n deadlineMs?: number\n /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */\n budgetUsd?: number\n /** Shared paid-call account when the analyst runs inside a larger campaign. */\n costLedger?: CostLedgerHandle\n /** Attribution phase used when writing to the shared paid-call account. */\n costPhase?: string\n /**\n * Shared chat client. Analysts that call an LLM go through this so\n * the operator picks transport (sandbox-sdk | router | cli-bridge |\n * direct-provider | mock) at the registry boundary without touching\n * analyst code.\n */\n chat?: ChatClient\n /**\n * Findings from a prior run the operator wants the analyst to see as\n * retrieval context. Kinds that take advantage of cross-run memory\n * (failure-mode \"I saw this cluster last run\", knowledge-gap \"the wiki\n * page I asked for is still missing\") render these into the actor's\n * working set. Filtering is the operator's job: pass the slice that\n * matches the analyst's id, or pass everything and let the kind\n * filter. Empty / absent means no cross-run context.\n */\n priorFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Findings emitted by analysts that completed earlier in this registry run.\n * This is separate from `priorFindings`: upstream findings are dependency\n * context for the current pass, while prior findings are cross-run memory.\n * The registry populates this only when `RegistryRunOpts.chainFindings` is on.\n */\n upstreamFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Report metered work independently of findings. This keeps an empty finding\n * set from erasing token/cost telemetry. Multiple receipts are accumulated.\n */\n recordUsage?: (receipt: AnalystUsageReceipt) => void\n /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */\n tags?: Record<string, string>\n /** Logger callback — analysts SHOULD prefer this over console.* for testability. */\n log?: (msg: string, fields?: Record<string, unknown>) => void\n /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */\n signal?: AbortSignal\n}\n\n/**\n * The minimal contract. Concrete analysts can refine `TInput` so\n * implementations stay type-safe (e.g. a trace analyst's `TInput` is\n * `TraceAnalysisStore`); the registry passes the right field from\n * `AnalystRunInputs` based on `inputKind`.\n */\nexport interface Analyst<TInput = unknown> {\n /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */\n readonly id: string\n /** Human-readable. One sentence. */\n readonly description: string\n readonly inputKind: AnalystInputKind\n readonly cost: AnalystCost\n readonly requires?: AnalystRequirements\n /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */\n readonly version: string\n analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>\n}\n\n/** Metered work performed by one analyst call. */\nexport interface AnalystUsageReceipt {\n /** Number of model-usage records observed at the provider boundary. */\n calls: number | null\n /** Null when the provider did not return token accounting. */\n tokens: RunTokenUsage | null\n /** Observed, estimated, or explicitly uncaptured dollar cost. */\n cost: RunCostProvenance\n /** Known lower bound when one or more calls have uncaptured cost. */\n knownCostUsd?: number\n /**\n * Token counts the provider reported on only one side. Present exactly when\n * `tokens` is null and at least one side WAS reported: `RunTokenUsage` has no\n * nullable side, so a one-sided count cannot live in `tokens` without writing\n * a zero nobody measured. Read it as a lower bound, never as a total — the\n * field exists so a null `tokens` cannot hide a real count.\n */\n partialTokens?: { input: number | null; output: number | null }\n /**\n * True when the token counts were DERIVED by the transport (from character\n * lengths, say) rather than measured by the model provider. `cost.kind` is\n * `estimated` both for a rate estimate over exact tokens and for one over\n * derived tokens; this is the field that separates them.\n */\n tokensEstimated?: boolean\n}\n\n// ── finding_id stability ─────────────────────────────────────────────\n\n/**\n * Compute the stable finding_id from the identity-defining fields.\n * Default implementation hashes {analyst_id, area, subject, normalized claim}.\n * Analysts that emit findings whose claim text varies per run (timestamps,\n * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,\n * or (b) move the variable part into `rationale`/`metadata` and keep the\n * `claim` static.\n */\nexport function computeFindingId(input: {\n analyst_id: string\n area: string\n subject?: string\n claim: string\n /** Override the claim for hashing — use when the displayed claim has run-specific bits. */\n id_basis?: string\n}): string {\n const basis = JSON.stringify({\n a: input.analyst_id,\n r: input.area,\n s: input.subject ?? '',\n c: normalizeClaim(input.id_basis ?? input.claim),\n })\n return `f_${createHash('sha256').update(basis).digest('hex').slice(0, 20)}`\n}\n\nfunction normalizeClaim(c: string): string {\n // Lowercase, collapse whitespace, strip trailing punctuation. Goal:\n // \"Leaf X failed install\" and \"Leaf X failed install.\" hash the same.\n return c\n .toLowerCase()\n .replace(/\\s+/g, ' ')\n .replace(/[.!?;:,]+$/g, '')\n .trim()\n}\n\n/**\n * Convenience factory: produce a fully-formed AnalystFinding with the\n * id computed automatically. Analyst code stays terse.\n */\nexport function makeFinding(\n init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): AnalystFinding {\n const { id_basis, produced_at, ...rest } = init\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({\n analyst_id: rest.analyst_id,\n area: rest.area,\n subject: rest.subject,\n claim: rest.claim,\n id_basis,\n }),\n produced_at: produced_at ?? new Date().toISOString(),\n ...rest,\n }\n}\n\n/** Build a finding whose source is explicitly allowed during candidate generation. */\nexport function makeProposalFinding(\n init: Omit<ProposalFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): ProposalFinding {\n const { proposal_origin, ...finding } = init\n return { ...makeFinding(finding), proposal_origin }\n}\n\n// ── Registry result envelope ────────────────────────────────────────\n\nexport interface AnalystRunSummary {\n analyst_id: string\n status: 'ok' | 'skipped' | 'failed'\n /** Why skipped — missing input, budget exceeded, capability unmet. */\n reason?: string\n findings_count: number\n latency_ms: number\n /** Additive model usage and cost provenance for this analyst. */\n usage: AnalystUsageReceipt\n /** When `status='failed'`: the error class + message, never the full stack. */\n error?: { class: string; message: string }\n}\n\nexport interface AnalystRunResult {\n run_id: string\n correlation_id: string\n started_at: string\n ended_at: string\n findings: AnalystFinding[]\n per_analyst: AnalystRunSummary[]\n /** Total LLM cost in USD across all analysts in this registry.run(). */\n total_cost_usd: number\n /**\n * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only\n * the known subtotal and must not be treated as the run's total spend.\n */\n total_cost_provenance?: RunCostProvenance\n}\n\n// ── Streaming event envelope ────────────────────────────────────────\n\n/**\n * Events emitted by `AnalystRegistry.runStream(...)` in real time as\n * the registry executes. UIs subscribe via `for await (const ev of\n * registry.runStream(...))`; `registry.run(...)` is a thin collector\n * over the same stream, so the two surfaces share their invariants.\n *\n * Per-finding events are intentionally omitted — analyzers are batch\n * operations (a recursive engine returns the full `findings:json[]` at the\n * end of the responder), so streaming inside one analyst would only\n * emit partial JSON consumers can't render. The kind-completion event\n * is the right granularity; subscribers wanting per-finding rendering\n * iterate `event.findings` themselves.\n */\nexport type AnalystRunEvent =\n | {\n type: 'run-started'\n run_id: string\n correlation_id: string\n started_at: string\n /** The ordered list of analyst ids the registry will run. */\n analyst_ids: ReadonlyArray<string>\n }\n | {\n type: 'analyst-skipped'\n summary: AnalystRunSummary\n }\n | {\n type: 'analyst-started'\n analyst_id: string\n started_at: string\n }\n | {\n type: 'analyst-completed'\n /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */\n summary: AnalystRunSummary\n findings: ReadonlyArray<AnalystFinding>\n }\n | {\n type: 'run-completed'\n result: AnalystRunResult\n }\n","import type { CostChannel, CostLedgerFilter, CostLedgerHandle } from '../cost-ledger'\nimport type { AnalystUsageReceipt } from './types'\n\nexport const DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS = 5_000\n\n/** Convert one ledger channel's complete call set into one analyst receipt. */\nexport function usageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n filter: CostChannel | CostLedgerFilter = 'analyst',\n): AnalystUsageReceipt {\n const resolvedFilter = typeof filter === 'string' ? { channel: filter } : filter\n const summary = ledger.summary(resolvedFilter)\n const receipts = ledger.list(resolvedFilter)\n const hasReasoningUsage = receipts.some((receipt) => receipt.reasoningTokens !== undefined)\n const hasCacheWriteUsage = receipts.some((receipt) => receipt.cacheWriteTokens !== undefined)\n const cost = summary.costProvenance\n return {\n calls: summary.totalCalls + summary.pendingCalls,\n tokens: summary.usageComplete\n ? {\n input: summary.inputTokens,\n output: summary.outputTokens,\n ...(hasReasoningUsage ? { reasoning: summary.reasoningTokens ?? 0 } : {}),\n ...(summary.cachedTokens > 0 ? { cached: summary.cachedTokens } : {}),\n ...(hasCacheWriteUsage ? { cacheWrite: summary.cacheWriteTokens ?? 0 } : {}),\n }\n : null,\n cost,\n ...(cost.kind === 'uncaptured' ? { knownCostUsd: summary.totalCostUsd } : {}),\n }\n}\n\nexport interface SettledUsageReceipt {\n settled: boolean\n pendingCalls: number\n receipt: AnalystUsageReceipt\n}\n\n/** Wait a bounded time for late provider receipts, then take one immutable snapshot. */\nexport async function settleUsageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n options: CostLedgerFilter & { timeoutMs?: number } = {},\n): Promise<SettledUsageReceipt> {\n const { timeoutMs: requestedTimeoutMs, ...requestedFilter } = options\n const filter: CostLedgerFilter = {\n channel: requestedFilter.channel ?? 'analyst',\n ...(requestedFilter.phase === undefined ? {} : { phase: requestedFilter.phase }),\n ...(requestedFilter.tags === undefined ? {} : { tags: requestedFilter.tags }),\n }\n const timeoutMs = validateUsageSettlementTimeout(requestedTimeoutMs)\n const initial = ledger.summary(filter)\n const waitResult =\n initial.pendingCalls === 0\n ? true\n : ledger.waitForIdle\n ? await ledger.waitForIdle({ timeoutMs })\n : false\n const pendingCalls = ledger.summary(filter).pendingCalls\n return {\n settled: waitResult && pendingCalls === 0,\n pendingCalls,\n receipt: usageReceiptFromCostLedger(ledger, filter),\n }\n}\n\nexport function validateUsageSettlementTimeout(timeoutMs?: number): number {\n const resolved = timeoutMs ?? DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS\n if (!Number.isSafeInteger(resolved) || resolved < 0 || resolved > 2_147_483_647) {\n throw new TypeError(\n 'settlementTimeoutMs must be a non-negative safe integer no greater than 2147483647',\n )\n }\n return resolved\n}\n\nexport function assertValidAnalystUsageReceipt(\n receipt: AnalystUsageReceipt,\n context = 'AnalystContext.recordUsage',\n): void {\n if (receipt.calls !== null && (!Number.isSafeInteger(receipt.calls) || receipt.calls < 0)) {\n throw new Error(`${context}: calls must be a non-negative safe integer or null`)\n }\n if (receipt.tokens) {\n assertNonNegativeSafeInteger(receipt.tokens.input, 'tokens.input', context)\n assertNonNegativeSafeInteger(receipt.tokens.output, 'tokens.output', context)\n if (receipt.tokens.reasoning !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.reasoning, 'tokens.reasoning', context)\n if (receipt.tokens.reasoning > receipt.tokens.output) {\n throw new Error(`${context}: tokens.reasoning must not exceed tokens.output`)\n }\n }\n if (receipt.tokens.cached !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cached, 'tokens.cached', context)\n }\n if (receipt.tokens.cacheWrite !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cacheWrite, 'tokens.cacheWrite', context)\n }\n }\n if (receipt.cost.kind !== 'uncaptured') {\n assertNonNegativeFinite(receipt.cost.usd, 'cost.usd', context)\n } else if (receipt.cost.usd !== null) {\n throw new Error(`${context}: uncaptured cost.usd must be null`)\n }\n if (receipt.knownCostUsd !== undefined) {\n assertNonNegativeFinite(receipt.knownCostUsd, 'knownCostUsd', context)\n }\n if (receipt.partialTokens) {\n const { input, output } = receipt.partialTokens\n if (receipt.tokens) {\n throw new Error(`${context}: partialTokens must be absent when tokens is complete`)\n }\n if (input === null && output === null) {\n throw new Error(`${context}: partialTokens must carry at least one reported side`)\n }\n if (input !== null) assertNonNegativeSafeInteger(input, 'partialTokens.input', context)\n if (output !== null) assertNonNegativeSafeInteger(output, 'partialTokens.output', context)\n }\n}\n\nfunction assertNonNegativeSafeInteger(value: number, field: string, context: string): void {\n if (!Number.isSafeInteger(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative safe integer`)\n }\n}\n\nfunction assertNonNegativeFinite(value: number, field: string, context: string): void {\n if (!Number.isFinite(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative finite number`)\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;AAiPA,SAAgB,iBAAiB,OAOtB;CACT,MAAM,QAAQ,KAAK,UAAU;EAC3B,GAAG,MAAM;EACT,GAAG,MAAM;EACT,GAAG,MAAM,WAAW;EACpB,GAAG,eAAe,MAAM,YAAY,MAAM,KAAK;CACjD,CAAC;CACD,OAAO,KAAK,WAAW,QAAQ,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1E;AAEA,SAAS,eAAe,GAAmB;CAGzC,OAAO,EACJ,YAAY,CAAC,CACb,QAAQ,QAAQ,GAAG,CAAC,CACpB,QAAQ,eAAe,EAAE,CAAC,CAC1B,KAAK;AACV;;;;;AAMA,SAAgB,YACd,MAIgB;CAChB,MAAM,EAAE,UAAU,aAAa,GAAG,SAAS;CAC3C,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAC3B,YAAY,KAAK;GACjB,MAAM,KAAK;GACX,SAAS,KAAK;GACd,OAAO,KAAK;GACZ;EACF,CAAC;EACD,aAAa,gCAAe,IAAI,KAAK,EAAA,CAAE,YAAY;EACnD,GAAG;CACL;AACF;;AAGA,SAAgB,oBACd,MAIiB;CACjB,MAAM,EAAE,iBAAiB,GAAG,YAAY;CACxC,OAAO;EAAE,GAAG,YAAY,OAAO;EAAG;CAAgB;AACpD;;ACxSA,SAAgB,2BACd,QACA,SAAyC,WACpB;CACrB,MAAM,iBAAiB,OAAO,WAAW,WAAW,EAAE,SAAS,OAAO,IAAI;CAC1E,MAAM,UAAU,OAAO,QAAQ,cAAc;CAC7C,MAAM,WAAW,OAAO,KAAK,cAAc;CAC3C,MAAM,oBAAoB,SAAS,MAAM,YAAY,QAAQ,oBAAoB,KAAA,CAAS;CAC1F,MAAM,qBAAqB,SAAS,MAAM,YAAY,QAAQ,qBAAqB,KAAA,CAAS;CAC5F,MAAM,OAAO,QAAQ;CACrB,OAAO;EACL,OAAO,QAAQ,aAAa,QAAQ;EACpC,QAAQ,QAAQ,gBACZ;GACE,OAAO,QAAQ;GACf,QAAQ,QAAQ;GAChB,GAAI,oBAAoB,EAAE,WAAW,QAAQ,mBAAmB,EAAE,IAAI,CAAC;GACvE,GAAI,QAAQ,eAAe,IAAI,EAAE,QAAQ,QAAQ,aAAa,IAAI,CAAC;GACnE,GAAI,qBAAqB,EAAE,YAAY,QAAQ,oBAAoB,EAAE,IAAI,CAAC;EAC5E,IACA;EACJ;EACA,GAAI,KAAK,SAAS,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CAC7E;AACF;;AASA,eAAsB,iCACpB,QACA,UAAqD,CAAC,GACxB;CAC9B,MAAM,EAAE,WAAW,oBAAoB,GAAG,oBAAoB;CAC9D,MAAM,SAA2B;EAC/B,SAAS,gBAAgB,WAAW;EACpC,GAAI,gBAAgB,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,gBAAgB,MAAM;EAC9E,GAAI,gBAAgB,SAAS,KAAA,IAAY,CAAC,IAAI,EAAE,MAAM,gBAAgB,KAAK;CAC7E;CACA,MAAM,YAAY,+BAA+B,kBAAkB;CAEnE,MAAM,aADU,OAAO,QAAQ,MAEvB,CAAC,CAAC,iBAAiB,IACrB,OACA,OAAO,cACL,MAAM,OAAO,YAAY,EAAE,UAAU,CAAC,IACtC;CACR,MAAM,eAAe,OAAO,QAAQ,MAAM,CAAC,CAAC;CAC5C,OAAO;EACL,SAAS,cAAc,iBAAiB;EACxC;EACA,SAAS,2BAA2B,QAAQ,MAAM;CACpD;AACF;AAEA,SAAgB,+BAA+B,WAA4B;CACzE,MAAM,WAAW,aAAA;CACjB,IAAI,CAAC,OAAO,cAAc,QAAQ,KAAK,WAAW,KAAK,WAAW,YAChE,MAAM,IAAI,UACR,oFACF;CAEF,OAAO;AACT;AAEA,SAAgB,+BACd,SACA,UAAU,8BACJ;CACN,IAAI,QAAQ,UAAU,SAAS,CAAC,OAAO,cAAc,QAAQ,KAAK,KAAK,QAAQ,QAAQ,IACrF,MAAM,IAAI,MAAM,GAAG,QAAQ,oDAAoD;CAEjF,IAAI,QAAQ,QAAQ;EAClB,6BAA6B,QAAQ,OAAO,OAAO,gBAAgB,OAAO;EAC1E,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAC5E,IAAI,QAAQ,OAAO,cAAc,KAAA,GAAW;GAC1C,6BAA6B,QAAQ,OAAO,WAAW,oBAAoB,OAAO;GAClF,IAAI,QAAQ,OAAO,YAAY,QAAQ,OAAO,QAC5C,MAAM,IAAI,MAAM,GAAG,QAAQ,iDAAiD;EAEhF;EACA,IAAI,QAAQ,OAAO,WAAW,KAAA,GAC5B,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAE9E,IAAI,QAAQ,OAAO,eAAe,KAAA,GAChC,6BAA6B,QAAQ,OAAO,YAAY,qBAAqB,OAAO;CAExF;CACA,IAAI,QAAQ,KAAK,SAAS,cACxB,wBAAwB,QAAQ,KAAK,KAAK,YAAY,OAAO;MACxD,IAAI,QAAQ,KAAK,QAAQ,MAC9B,MAAM,IAAI,MAAM,GAAG,QAAQ,mCAAmC;CAEhE,IAAI,QAAQ,iBAAiB,KAAA,GAC3B,wBAAwB,QAAQ,cAAc,gBAAgB,OAAO;CAEvE,IAAI,QAAQ,eAAe;EACzB,MAAM,EAAE,OAAO,WAAW,QAAQ;EAClC,IAAI,QAAQ,QACV,MAAM,IAAI,MAAM,GAAG,QAAQ,uDAAuD;EAEpF,IAAI,UAAU,QAAQ,WAAW,MAC/B,MAAM,IAAI,MAAM,GAAG,QAAQ,sDAAsD;EAEnF,IAAI,UAAU,MAAM,6BAA6B,OAAO,uBAAuB,OAAO;EACtF,IAAI,WAAW,MAAM,6BAA6B,QAAQ,wBAAwB,OAAO;CAC3F;AACF;AAEA,SAAS,6BAA6B,OAAe,OAAe,SAAuB;CACzF,IAAI,CAAC,OAAO,cAAc,KAAK,KAAK,QAAQ,GAC1C,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,qCAAqC;AAE9E;AAEA,SAAS,wBAAwB,OAAe,OAAe,SAAuB;CACpF,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,QAAQ,GACrC,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,sCAAsC;AAE/E"}
|
|
1
|
+
{"version":3,"file":"usage-receipt-EVI8B8Xu.js","names":[],"sources":["../src/analyst/types.ts","../src/analyst/usage-receipt.ts"],"sourcesContent":["/**\n * Analyst contract — the missing orchestration layer over agent-eval's\n * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,\n * SemanticConceptJudge, JudgeFn, ...).\n *\n * Each existing primitive returns its own output shape. The Analyst\n * contract is the single envelope every primitive lifts into, so a\n * registry can run N analysts against a run and a single renderer can\n * compose findings without knowing which analyzer produced them.\n *\n * The contract is intentionally domain-agnostic: nothing here knows\n * about code, voice, RAG, or any particular agent stack. Analysts\n * declare what INPUT KIND they need (a trace store, an artifact dir,\n * a RunRecord, a JudgeInput, or `custom`), and the registry routes\n * the matching input from `AnalystRunInputs`.\n */\n\nimport { createHash } from 'node:crypto'\nimport type { CostLedgerHandle } from '../cost-ledger'\nimport type { RunCostProvenance, RunRecord, RunTokenUsage } from '../run-record'\nimport type { TraceAnalysisStore } from '../trace-analyst/store'\nimport type { JudgeInput } from '../types'\nimport type { ChatClient } from './chat-client'\n\n/**\n * Unified envelope every analyst emits. Schema-versioned so renderers\n * and time-series diffs survive future field additions.\n */\nexport interface AnalystFinding {\n schema_version: '1.0.0'\n /**\n * Stable hash over identity-defining fields (analyst_id + canonical\n * claim + area + optional subject). Two findings from two runs that\n * \"are the same finding\" share this id — that's what `diffFindings`\n * uses to compute appeared/disappeared sets across runs.\n */\n finding_id: string\n analyst_id: string\n produced_at: string\n severity: AnalystSeverity\n /**\n * Coarse classification. Renderers group by this. Free-form so\n * domain-specific analysts can introduce categories without a\n * schema change ('agent-reasoning', 'verification', 'cost',\n * 'tool-use', 'safety', 'latency', 'data-quality', ...).\n */\n area: string\n claim: string\n rationale?: string\n evidence_refs: EvidenceRef[]\n recommended_action?: string\n validation_plan?: string\n /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */\n confidence: number\n /**\n * Optional subject the finding is about — leaf id, agent id, request\n * id. Included in finding_id when present so per-subject findings\n * diff cleanly across runs.\n */\n subject?: string\n /** True when this finding was lifted from a judge result rather than observed\n * directly in a trace or artifact. Descriptive only: proposal access is\n * controlled by `ProposalFinding.proposal_origin`. */\n derived_from_judge?: boolean\n /** Analyst-private extras; renderers ignore unless they know the analyst. */\n metadata?: Record<string, unknown>\n}\n\nexport type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info'\n\n/** Data sources that candidate generation may intentionally learn from. */\nexport type ProposalFindingOrigin = 'search' | 'production'\n\n/** A finding explicitly admitted as candidate-generation input. */\nexport type ProposalFinding = AnalystFinding & {\n readonly proposal_origin: ProposalFindingOrigin\n}\n\nexport interface EvidenceRef {\n /**\n * Where the evidence lives. `span` and `event` refer to OTLP trace\n * elements; `artifact` to a file inside the run's artifact tree;\n * `finding` to another AnalystFinding (cross-analyst chaining);\n * `metric` to a named scalar reading the renderer knows how to read.\n */\n kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric'\n uri: string\n excerpt?: string\n}\n\n// ── Analyst contract ─────────────────────────────────────────────────\n\n/**\n * The discriminator the registry uses to pass the right input.\n * `custom` is the escape hatch — analysts that need something else\n * (e.g. an embedding cache, a partner SDK handle) read it from\n * `AnalystRunInputs.custom[<analyst id>]`.\n */\nexport type AnalystInputKind =\n | 'trace-store'\n | 'artifact-dir'\n | 'run-record'\n | 'judge-input'\n | 'custom'\n\nexport interface AnalystCost {\n /** `deterministic` analysts MUST NOT call the LLM. */\n kind: 'deterministic' | 'llm'\n /** Optional declared upper bound; the registry can enforce a budget. */\n est_usd_per_run?: number\n /** Models the analyst expects to use (informational). */\n models?: string[]\n /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */\n settlement_timeout_ms?: number\n}\n\nexport interface AnalystRequirements {\n /** Min number of shots / samples the analyst needs to produce signal. */\n min_shots?: number\n /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */\n capabilities?: string[]\n}\n\n/**\n * What's passed to every analyst call. The registry resolves which\n * field the analyst's `inputKind` selects and asserts it's present.\n */\nexport interface AnalystRunInputs {\n traceStore?: TraceAnalysisStore\n artifactDir?: string\n runRecord?: RunRecord\n judgeInput?: JudgeInput\n /** Keyed by analyst id; populated by callers that registered custom analysts. */\n custom?: Record<string, unknown>\n}\n\nexport interface AnalystContext {\n runId: string\n /** Stable correlation id so logs from a single registry.run() share a tag. */\n correlationId: string\n /** Enforced wall-clock deadline (epoch ms). */\n deadlineMs?: number\n /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */\n budgetUsd?: number\n /** Shared paid-call account when the analyst runs inside a larger campaign. */\n costLedger?: CostLedgerHandle\n /** Attribution phase used when writing to the shared paid-call account. */\n costPhase?: string\n /**\n * Shared chat client. Analysts that call an LLM go through this so\n * the operator picks transport (sandbox-sdk | router | cli-bridge |\n * direct-provider | mock) at the registry boundary without touching\n * analyst code.\n */\n chat?: ChatClient\n /**\n * Findings from a prior run the operator wants the analyst to see as\n * retrieval context. Kinds that take advantage of cross-run memory\n * (failure-mode \"I saw this cluster last run\", knowledge-gap \"the wiki\n * page I asked for is still missing\") render these into the actor's\n * working set. Filtering is the operator's job: pass the slice that\n * matches the analyst's id, or pass everything and let the kind\n * filter. Empty / absent means no cross-run context.\n */\n priorFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Findings emitted by analysts that completed earlier in this registry run.\n * This is separate from `priorFindings`: upstream findings are dependency\n * context for the current pass, while prior findings are cross-run memory.\n * The registry populates this only when `RegistryRunOpts.chainFindings` is on.\n */\n upstreamFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Report metered work independently of findings. This keeps an empty finding\n * set from erasing token/cost telemetry. Multiple receipts are accumulated.\n */\n recordUsage?: (receipt: AnalystUsageReceipt) => void\n /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */\n tags?: Record<string, string>\n /** Logger callback — analysts SHOULD prefer this over console.* for testability. */\n log?: (msg: string, fields?: Record<string, unknown>) => void\n /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */\n signal?: AbortSignal\n /**\n * Optional live-execution port. A runtime that owns a sandbox or checkout\n * fills it so an analyst can execute a bounded probe against the run's\n * produced state instead of reasoning about it from the trace alone. This\n * package defines only the port: no field here reaches for an agent loop,\n * and an absent probe means the analyst works from recorded evidence.\n */\n probe?: ExecutionProbe\n}\n\n// ── Live-execution port ─────────────────────────────────────────────\n\n/** One bounded command an analyst asks the probe to run. */\nexport interface ExecutionProbeRequest {\n command: string\n /** Working directory inside the probed environment. */\n cwd?: string\n /** Hard wall-clock deadline for this one execution. */\n timeoutMs: number\n /** Bytes of combined output retained; the prober truncates beyond it. */\n maxOutputBytes?: number\n signal?: AbortSignal\n}\n\n/**\n * Typed outcome of one probe execution. `succeeded: false` is a PROBE failure\n * (the environment could not run the command); a command that ran and exited\n * non-zero is a successful observation with a non-zero `exitCode`.\n */\nexport type ExecutionProbeOutcome =\n | {\n succeeded: true\n exitCode: number\n stdout: string\n stderr: string\n durationMs: number\n /** True when output was cut at `maxOutputBytes`. */\n truncated: boolean\n }\n | { succeeded: false; error: { class: string; message: string } }\n\n/**\n * The seam a runtime fills to let analysts observe produced state live.\n * Implementations own sandboxing, credentials, and cleanup; analysts only\n * submit bounded requests and read typed outcomes.\n */\nexport interface ExecutionProbe {\n /** One plain sentence naming what is being probed (e.g. a sandbox id). */\n readonly description: string\n execute(request: ExecutionProbeRequest): Promise<ExecutionProbeOutcome>\n}\n\n/**\n * The minimal contract. Concrete analysts can refine `TInput` so\n * implementations stay type-safe (e.g. a trace analyst's `TInput` is\n * `TraceAnalysisStore`); the registry passes the right field from\n * `AnalystRunInputs` based on `inputKind`.\n */\nexport interface Analyst<TInput = unknown> {\n /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */\n readonly id: string\n /** Human-readable. One sentence. */\n readonly description: string\n readonly inputKind: AnalystInputKind\n readonly cost: AnalystCost\n readonly requires?: AnalystRequirements\n /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */\n readonly version: string\n analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>\n}\n\n/** Metered work performed by one analyst call. */\nexport interface AnalystUsageReceipt {\n /** Number of model-usage records observed at the provider boundary. */\n calls: number | null\n /** Null when the provider did not return token accounting. */\n tokens: RunTokenUsage | null\n /** Observed, estimated, or explicitly uncaptured dollar cost. */\n cost: RunCostProvenance\n /** Known lower bound when one or more calls have uncaptured cost. */\n knownCostUsd?: number\n /**\n * Token counts the provider reported on only one side. Present exactly when\n * `tokens` is null and at least one side WAS reported: `RunTokenUsage` has no\n * nullable side, so a one-sided count cannot live in `tokens` without writing\n * a zero nobody measured. Read it as a lower bound, never as a total — the\n * field exists so a null `tokens` cannot hide a real count.\n */\n partialTokens?: { input: number | null; output: number | null }\n /**\n * True when the token counts were DERIVED by the transport (from character\n * lengths, say) rather than measured by the model provider. `cost.kind` is\n * `estimated` both for a rate estimate over exact tokens and for one over\n * derived tokens; this is the field that separates them.\n */\n tokensEstimated?: boolean\n}\n\n// ── finding_id stability ─────────────────────────────────────────────\n\n/**\n * Compute the stable finding_id from the identity-defining fields.\n * Default implementation hashes {analyst_id, area, subject, normalized claim}.\n * Analysts that emit findings whose claim text varies per run (timestamps,\n * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,\n * or (b) move the variable part into `rationale`/`metadata` and keep the\n * `claim` static.\n */\nexport function computeFindingId(input: {\n analyst_id: string\n area: string\n subject?: string\n claim: string\n /** Override the claim for hashing — use when the displayed claim has run-specific bits. */\n id_basis?: string\n}): string {\n const basis = JSON.stringify({\n a: input.analyst_id,\n r: input.area,\n s: input.subject ?? '',\n c: normalizeClaim(input.id_basis ?? input.claim),\n })\n return `f_${createHash('sha256').update(basis).digest('hex').slice(0, 20)}`\n}\n\nfunction normalizeClaim(c: string): string {\n // Lowercase, collapse whitespace, strip trailing punctuation. Goal:\n // \"Leaf X failed install\" and \"Leaf X failed install.\" hash the same.\n return c\n .toLowerCase()\n .replace(/\\s+/g, ' ')\n .replace(/[.!?;:,]+$/g, '')\n .trim()\n}\n\n/**\n * Convenience factory: produce a fully-formed AnalystFinding with the\n * id computed automatically. Analyst code stays terse.\n */\nexport function makeFinding(\n init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): AnalystFinding {\n const { id_basis, produced_at, ...rest } = init\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({\n analyst_id: rest.analyst_id,\n area: rest.area,\n subject: rest.subject,\n claim: rest.claim,\n id_basis,\n }),\n produced_at: produced_at ?? new Date().toISOString(),\n ...rest,\n }\n}\n\n/** Build a finding whose source is explicitly allowed during candidate generation. */\nexport function makeProposalFinding(\n init: Omit<ProposalFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): ProposalFinding {\n const { proposal_origin, ...finding } = init\n return { ...makeFinding(finding), proposal_origin }\n}\n\n// ── Registry result envelope ────────────────────────────────────────\n\nexport interface AnalystRunSummary {\n analyst_id: string\n status: 'ok' | 'skipped' | 'failed'\n /** Why skipped — missing input, budget exceeded, capability unmet. */\n reason?: string\n findings_count: number\n latency_ms: number\n /** Additive model usage and cost provenance for this analyst. */\n usage: AnalystUsageReceipt\n /** When `status='failed'`: the error class + message, never the full stack. */\n error?: { class: string; message: string }\n}\n\nexport interface AnalystRunResult {\n run_id: string\n correlation_id: string\n started_at: string\n ended_at: string\n findings: AnalystFinding[]\n per_analyst: AnalystRunSummary[]\n /** Total LLM cost in USD across all analysts in this registry.run(). */\n total_cost_usd: number\n /**\n * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only\n * the known subtotal and must not be treated as the run's total spend.\n */\n total_cost_provenance?: RunCostProvenance\n}\n\n// ── Streaming event envelope ────────────────────────────────────────\n\n/**\n * Events emitted by `AnalystRegistry.runStream(...)` in real time as\n * the registry executes. UIs subscribe via `for await (const ev of\n * registry.runStream(...))`; `registry.run(...)` is a thin collector\n * over the same stream, so the two surfaces share their invariants.\n *\n * Per-finding events are intentionally omitted — analyzers are batch\n * operations (a recursive engine returns the full `findings:json[]` at the\n * end of the responder), so streaming inside one analyst would only\n * emit partial JSON consumers can't render. The kind-completion event\n * is the right granularity; subscribers wanting per-finding rendering\n * iterate `event.findings` themselves.\n */\nexport type AnalystRunEvent =\n | {\n type: 'run-started'\n run_id: string\n correlation_id: string\n started_at: string\n /** The ordered list of analyst ids the registry will run. */\n analyst_ids: ReadonlyArray<string>\n }\n | {\n type: 'analyst-skipped'\n summary: AnalystRunSummary\n }\n | {\n type: 'analyst-started'\n analyst_id: string\n started_at: string\n }\n | {\n type: 'analyst-completed'\n /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */\n summary: AnalystRunSummary\n findings: ReadonlyArray<AnalystFinding>\n }\n | {\n type: 'run-completed'\n result: AnalystRunResult\n }\n","import type { CostChannel, CostLedgerFilter, CostLedgerHandle } from '../cost-ledger'\nimport type { AnalystUsageReceipt } from './types'\n\nexport const DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS = 5_000\n\n/** Convert one ledger channel's complete call set into one analyst receipt. */\nexport function usageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n filter: CostChannel | CostLedgerFilter = 'analyst',\n): AnalystUsageReceipt {\n const resolvedFilter = typeof filter === 'string' ? { channel: filter } : filter\n const summary = ledger.summary(resolvedFilter)\n const receipts = ledger.list(resolvedFilter)\n const hasReasoningUsage = receipts.some((receipt) => receipt.reasoningTokens !== undefined)\n const hasCacheWriteUsage = receipts.some((receipt) => receipt.cacheWriteTokens !== undefined)\n const cost = summary.costProvenance\n return {\n calls: summary.totalCalls + summary.pendingCalls,\n tokens: summary.usageComplete\n ? {\n input: summary.inputTokens,\n output: summary.outputTokens,\n ...(hasReasoningUsage ? { reasoning: summary.reasoningTokens ?? 0 } : {}),\n ...(summary.cachedTokens > 0 ? { cached: summary.cachedTokens } : {}),\n ...(hasCacheWriteUsage ? { cacheWrite: summary.cacheWriteTokens ?? 0 } : {}),\n }\n : null,\n cost,\n ...(cost.kind === 'uncaptured' ? { knownCostUsd: summary.totalCostUsd } : {}),\n }\n}\n\nexport interface SettledUsageReceipt {\n settled: boolean\n pendingCalls: number\n receipt: AnalystUsageReceipt\n}\n\n/** Wait a bounded time for late provider receipts, then take one immutable snapshot. */\nexport async function settleUsageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n options: CostLedgerFilter & { timeoutMs?: number } = {},\n): Promise<SettledUsageReceipt> {\n const { timeoutMs: requestedTimeoutMs, ...requestedFilter } = options\n const filter: CostLedgerFilter = {\n channel: requestedFilter.channel ?? 'analyst',\n ...(requestedFilter.phase === undefined ? {} : { phase: requestedFilter.phase }),\n ...(requestedFilter.tags === undefined ? {} : { tags: requestedFilter.tags }),\n }\n const timeoutMs = validateUsageSettlementTimeout(requestedTimeoutMs)\n const initial = ledger.summary(filter)\n const waitResult =\n initial.pendingCalls === 0\n ? true\n : ledger.waitForIdle\n ? await ledger.waitForIdle({ timeoutMs })\n : false\n const pendingCalls = ledger.summary(filter).pendingCalls\n return {\n settled: waitResult && pendingCalls === 0,\n pendingCalls,\n receipt: usageReceiptFromCostLedger(ledger, filter),\n }\n}\n\nexport function validateUsageSettlementTimeout(timeoutMs?: number): number {\n const resolved = timeoutMs ?? DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS\n if (!Number.isSafeInteger(resolved) || resolved < 0 || resolved > 2_147_483_647) {\n throw new TypeError(\n 'settlementTimeoutMs must be a non-negative safe integer no greater than 2147483647',\n )\n }\n return resolved\n}\n\nexport function assertValidAnalystUsageReceipt(\n receipt: AnalystUsageReceipt,\n context = 'AnalystContext.recordUsage',\n): void {\n if (receipt.calls !== null && (!Number.isSafeInteger(receipt.calls) || receipt.calls < 0)) {\n throw new Error(`${context}: calls must be a non-negative safe integer or null`)\n }\n if (receipt.tokens) {\n assertNonNegativeSafeInteger(receipt.tokens.input, 'tokens.input', context)\n assertNonNegativeSafeInteger(receipt.tokens.output, 'tokens.output', context)\n if (receipt.tokens.reasoning !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.reasoning, 'tokens.reasoning', context)\n if (receipt.tokens.reasoning > receipt.tokens.output) {\n throw new Error(`${context}: tokens.reasoning must not exceed tokens.output`)\n }\n }\n if (receipt.tokens.cached !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cached, 'tokens.cached', context)\n }\n if (receipt.tokens.cacheWrite !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cacheWrite, 'tokens.cacheWrite', context)\n }\n }\n if (receipt.cost.kind !== 'uncaptured') {\n assertNonNegativeFinite(receipt.cost.usd, 'cost.usd', context)\n } else if (receipt.cost.usd !== null) {\n throw new Error(`${context}: uncaptured cost.usd must be null`)\n }\n if (receipt.knownCostUsd !== undefined) {\n assertNonNegativeFinite(receipt.knownCostUsd, 'knownCostUsd', context)\n }\n if (receipt.partialTokens) {\n const { input, output } = receipt.partialTokens\n if (receipt.tokens) {\n throw new Error(`${context}: partialTokens must be absent when tokens is complete`)\n }\n if (input === null && output === null) {\n throw new Error(`${context}: partialTokens must carry at least one reported side`)\n }\n if (input !== null) assertNonNegativeSafeInteger(input, 'partialTokens.input', context)\n if (output !== null) assertNonNegativeSafeInteger(output, 'partialTokens.output', context)\n }\n}\n\nfunction assertNonNegativeSafeInteger(value: number, field: string, context: string): void {\n if (!Number.isSafeInteger(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative safe integer`)\n }\n}\n\nfunction assertNonNegativeFinite(value: number, field: string, context: string): void {\n if (!Number.isFinite(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative finite number`)\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;AAmSA,SAAgB,iBAAiB,OAOtB;CACT,MAAM,QAAQ,KAAK,UAAU;EAC3B,GAAG,MAAM;EACT,GAAG,MAAM;EACT,GAAG,MAAM,WAAW;EACpB,GAAG,eAAe,MAAM,YAAY,MAAM,KAAK;CACjD,CAAC;CACD,OAAO,KAAK,WAAW,QAAQ,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1E;AAEA,SAAS,eAAe,GAAmB;CAGzC,OAAO,EACJ,YAAY,CAAC,CACb,QAAQ,QAAQ,GAAG,CAAC,CACpB,QAAQ,eAAe,EAAE,CAAC,CAC1B,KAAK;AACV;;;;;AAMA,SAAgB,YACd,MAIgB;CAChB,MAAM,EAAE,UAAU,aAAa,GAAG,SAAS;CAC3C,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAC3B,YAAY,KAAK;GACjB,MAAM,KAAK;GACX,SAAS,KAAK;GACd,OAAO,KAAK;GACZ;EACF,CAAC;EACD,aAAa,gCAAe,IAAI,KAAK,EAAA,CAAE,YAAY;EACnD,GAAG;CACL;AACF;;AAGA,SAAgB,oBACd,MAIiB;CACjB,MAAM,EAAE,iBAAiB,GAAG,YAAY;CACxC,OAAO;EAAE,GAAG,YAAY,OAAO;EAAG;CAAgB;AACpD;;AC1VA,SAAgB,2BACd,QACA,SAAyC,WACpB;CACrB,MAAM,iBAAiB,OAAO,WAAW,WAAW,EAAE,SAAS,OAAO,IAAI;CAC1E,MAAM,UAAU,OAAO,QAAQ,cAAc;CAC7C,MAAM,WAAW,OAAO,KAAK,cAAc;CAC3C,MAAM,oBAAoB,SAAS,MAAM,YAAY,QAAQ,oBAAoB,KAAA,CAAS;CAC1F,MAAM,qBAAqB,SAAS,MAAM,YAAY,QAAQ,qBAAqB,KAAA,CAAS;CAC5F,MAAM,OAAO,QAAQ;CACrB,OAAO;EACL,OAAO,QAAQ,aAAa,QAAQ;EACpC,QAAQ,QAAQ,gBACZ;GACE,OAAO,QAAQ;GACf,QAAQ,QAAQ;GAChB,GAAI,oBAAoB,EAAE,WAAW,QAAQ,mBAAmB,EAAE,IAAI,CAAC;GACvE,GAAI,QAAQ,eAAe,IAAI,EAAE,QAAQ,QAAQ,aAAa,IAAI,CAAC;GACnE,GAAI,qBAAqB,EAAE,YAAY,QAAQ,oBAAoB,EAAE,IAAI,CAAC;EAC5E,IACA;EACJ;EACA,GAAI,KAAK,SAAS,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CAC7E;AACF;;AASA,eAAsB,iCACpB,QACA,UAAqD,CAAC,GACxB;CAC9B,MAAM,EAAE,WAAW,oBAAoB,GAAG,oBAAoB;CAC9D,MAAM,SAA2B;EAC/B,SAAS,gBAAgB,WAAW;EACpC,GAAI,gBAAgB,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,gBAAgB,MAAM;EAC9E,GAAI,gBAAgB,SAAS,KAAA,IAAY,CAAC,IAAI,EAAE,MAAM,gBAAgB,KAAK;CAC7E;CACA,MAAM,YAAY,+BAA+B,kBAAkB;CAEnE,MAAM,aADU,OAAO,QAAQ,MAEvB,CAAC,CAAC,iBAAiB,IACrB,OACA,OAAO,cACL,MAAM,OAAO,YAAY,EAAE,UAAU,CAAC,IACtC;CACR,MAAM,eAAe,OAAO,QAAQ,MAAM,CAAC,CAAC;CAC5C,OAAO;EACL,SAAS,cAAc,iBAAiB;EACxC;EACA,SAAS,2BAA2B,QAAQ,MAAM;CACpD;AACF;AAEA,SAAgB,+BAA+B,WAA4B;CACzE,MAAM,WAAW,aAAA;CACjB,IAAI,CAAC,OAAO,cAAc,QAAQ,KAAK,WAAW,KAAK,WAAW,YAChE,MAAM,IAAI,UACR,oFACF;CAEF,OAAO;AACT;AAEA,SAAgB,+BACd,SACA,UAAU,8BACJ;CACN,IAAI,QAAQ,UAAU,SAAS,CAAC,OAAO,cAAc,QAAQ,KAAK,KAAK,QAAQ,QAAQ,IACrF,MAAM,IAAI,MAAM,GAAG,QAAQ,oDAAoD;CAEjF,IAAI,QAAQ,QAAQ;EAClB,6BAA6B,QAAQ,OAAO,OAAO,gBAAgB,OAAO;EAC1E,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAC5E,IAAI,QAAQ,OAAO,cAAc,KAAA,GAAW;GAC1C,6BAA6B,QAAQ,OAAO,WAAW,oBAAoB,OAAO;GAClF,IAAI,QAAQ,OAAO,YAAY,QAAQ,OAAO,QAC5C,MAAM,IAAI,MAAM,GAAG,QAAQ,iDAAiD;EAEhF;EACA,IAAI,QAAQ,OAAO,WAAW,KAAA,GAC5B,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAE9E,IAAI,QAAQ,OAAO,eAAe,KAAA,GAChC,6BAA6B,QAAQ,OAAO,YAAY,qBAAqB,OAAO;CAExF;CACA,IAAI,QAAQ,KAAK,SAAS,cACxB,wBAAwB,QAAQ,KAAK,KAAK,YAAY,OAAO;MACxD,IAAI,QAAQ,KAAK,QAAQ,MAC9B,MAAM,IAAI,MAAM,GAAG,QAAQ,mCAAmC;CAEhE,IAAI,QAAQ,iBAAiB,KAAA,GAC3B,wBAAwB,QAAQ,cAAc,gBAAgB,OAAO;CAEvE,IAAI,QAAQ,eAAe;EACzB,MAAM,EAAE,OAAO,WAAW,QAAQ;EAClC,IAAI,QAAQ,QACV,MAAM,IAAI,MAAM,GAAG,QAAQ,uDAAuD;EAEpF,IAAI,UAAU,QAAQ,WAAW,MAC/B,MAAM,IAAI,MAAM,GAAG,QAAQ,sDAAsD;EAEnF,IAAI,UAAU,MAAM,6BAA6B,OAAO,uBAAuB,OAAO;EACtF,IAAI,WAAW,MAAM,6BAA6B,QAAQ,wBAAwB,OAAO;CAC3F;AACF;AAEA,SAAS,6BAA6B,OAAe,OAAe,SAAuB;CACzF,IAAI,CAAC,OAAO,cAAc,KAAK,KAAK,QAAQ,GAC1C,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,qCAAqC;AAE9E;AAEA,SAAS,wBAAwB,OAAe,OAAe,SAAuB;CACpF,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,QAAQ,GACrC,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,sCAAsC;AAE/E"}
|