@tangle-network/agent-eval 0.144.11 → 0.144.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/dist/agent-profile-B9_GGsG8.d.ts +84 -0
- package/dist/agent-profile-B9_GGsG8.d.ts.map +1 -0
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts → agent-profile-cell-BkcRDikH.d.ts} +2 -2
- package/dist/{agent-profile-cell-BOP-iA9Q.d.ts.map → agent-profile-cell-BkcRDikH.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +134 -16
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +364 -10
- package/dist/analyst/index.js.map +1 -1
- package/dist/backend-integrity-CsVin_Wb.d.ts +280 -0
- package/dist/backend-integrity-CsVin_Wb.d.ts.map +1 -0
- package/dist/{baseline-C-GocmIW.js → baseline-BhPRQBVn.js} +3 -2
- package/dist/{baseline-C-GocmIW.js.map → baseline-BhPRQBVn.js.map} +1 -1
- package/dist/{benchmark-CWeqGl7x.js → benchmark-BhT16ep9.js} +2 -2
- package/dist/{benchmark-CWeqGl7x.js.map → benchmark-BhT16ep9.js.map} +1 -1
- package/dist/{agent-profile-DPi7IZg7.d.ts → benchmark-Ceoan7vk.d.ts} +4 -91
- package/dist/benchmark-Ceoan7vk.d.ts.map +1 -0
- package/dist/{benchmark-command-BteMFN62.js → benchmark-command-CY6cbhOB.js} +36 -79
- package/dist/benchmark-command-CY6cbhOB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +244 -2
- package/dist/benchmarks/index.d.ts.map +1 -0
- package/dist/benchmarks/index.js +733 -1
- package/dist/benchmarks/index.js.map +1 -0
- package/dist/builder-eval/index.d.ts +23 -2
- package/dist/builder-eval/index.d.ts.map +1 -1
- package/dist/builder-eval/index.js +227 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +10 -8
- package/dist/campaign/index.js +9 -6
- package/dist/{campaign-C2TTzQII.js → campaign-BYjBAypg.js} +21 -634
- package/dist/campaign-BYjBAypg.js.map +1 -0
- package/dist/{canonical-D011XM8r.js → canonical-D-XsTQ6_.js} +2 -2
- package/dist/{canonical-D011XM8r.js.map → canonical-D-XsTQ6_.js.map} +1 -1
- package/dist/{index-CQTZ-4XN.d.ts → capture-fetch-DDvpjVRU.d.ts} +3 -3
- package/dist/capture-fetch-DDvpjVRU.d.ts.map +1 -0
- package/dist/{default-registry-BmktKy8r.js → chat-client-Bp4Ebuuc.js} +1276 -1276
- package/dist/chat-client-Bp4Ebuuc.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-C9gzZE59.d.ts → client-KUbGulm_.d.ts} +4 -4
- package/dist/{client-C9gzZE59.d.ts.map → client-KUbGulm_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -390
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +18 -542
- package/dist/contract/index.js.map +1 -1
- package/dist/{cost-ledger-DMFxsLKr.js → cost-ledger-BSe92yAV.js} +3 -3
- package/dist/{cost-ledger-DMFxsLKr.js.map → cost-ledger-BSe92yAV.js.map} +1 -1
- package/dist/{cost-ledger-Bv_e8XHY.d.ts → cost-ledger-DbQdN3nO.d.ts} +2 -2
- package/dist/{cost-ledger-Bv_e8XHY.d.ts.map → cost-ledger-DbQdN3nO.d.ts.map} +1 -1
- package/dist/{counterfactual-CxmxAONP.d.ts → counterfactual--bpysZF0.d.ts} +2 -14
- package/dist/{counterfactual-CxmxAONP.d.ts.map → counterfactual--bpysZF0.d.ts.map} +1 -1
- package/dist/{counterfactual-CWPTrMH7.js → counterfactual-lDfCx0Uz.js} +3 -31
- package/dist/{counterfactual-CWPTrMH7.js.map → counterfactual-lDfCx0Uz.js.map} +1 -1
- package/dist/{dataset-C8xaLXdY.d.ts → dataset-CJjKqQfA.d.ts} +2 -8
- package/dist/dataset-CJjKqQfA.d.ts.map +1 -0
- package/dist/{default-registry-Bf8Woqmq.d.ts → default-registry-Di6HP6pG.d.ts} +7 -12
- package/dist/default-registry-Di6HP6pG.d.ts.map +1 -0
- package/dist/{analyze-runs-C30yljDJ.js → define-agent-eval-iqjT--ZZ.js} +542 -130
- package/dist/define-agent-eval-iqjT--ZZ.js.map +1 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts +388 -0
- package/dist/define-agent-eval-sH24zBfM.d.ts.map +1 -0
- package/dist/descriptive-B5MwKfbf.js +144 -0
- package/dist/descriptive-B5MwKfbf.js.map +1 -0
- package/dist/{dspy-rlm-engine-DbTk4JdR.js → dspy-rlm-engine-B_qhSc21.js} +4 -4
- package/dist/{dspy-rlm-engine-DbTk4JdR.js.map → dspy-rlm-engine-B_qhSc21.js.map} +1 -1
- package/dist/effect-sizes-DiH8MGOH.js +82 -0
- package/dist/effect-sizes-DiH8MGOH.js.map +1 -0
- package/dist/{engine-3hL-XqwJ.d.ts → engine-otFpE2gF.d.ts} +10 -38
- package/dist/engine-otFpE2gF.d.ts.map +1 -0
- package/dist/{errors-CKPfb2aH.d.ts → errors-DEE6u6ot.d.ts} +2 -14
- package/dist/{errors-CKPfb2aH.d.ts.map → errors-DEE6u6ot.d.ts.map} +1 -1
- package/dist/{errors-D-LKuDhb.js → errors-Dngq5h35.js} +2 -8
- package/dist/{errors-D-LKuDhb.js.map → errors-Dngq5h35.js.map} +1 -1
- package/dist/{eval-campaign-B_7wcnav.js → eval-campaign-UB-usSQ2.js} +6 -6
- package/dist/{eval-campaign-B_7wcnav.js.map → eval-campaign-UB-usSQ2.js.map} +1 -1
- package/dist/{exact-types-DSFFpLLI.d.ts → exact-types-BH1twmAJ.d.ts} +2 -2
- package/dist/{exact-types-DSFFpLLI.d.ts.map → exact-types-BH1twmAJ.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +9 -6
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +11 -7
- package/dist/experiment/index.js.map +1 -1
- package/dist/experiment-tracker-C29gXM4B.js +269 -0
- package/dist/experiment-tracker-C29gXM4B.js.map +1 -0
- package/dist/{experiment-tracker-IMntXr6J.d.ts → experiment-tracker-DWHZBAYL.d.ts} +77 -121
- package/dist/experiment-tracker-DWHZBAYL.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts → external-optimizer-contracts-lixrOZdX.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CZuJNcT5.d.ts.map → external-optimizer-contracts-lixrOZdX.d.ts.map} +1 -1
- package/dist/external-optimizer-process-BTiNB-RH.js +301 -0
- package/dist/external-optimizer-process-BTiNB-RH.js.map +1 -0
- package/dist/{single-run-lock-DFWHEB09.js → external-optimizer-subprocess-DrJ9hR8u.js} +144 -438
- package/dist/external-optimizer-subprocess-DrJ9hR8u.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/{failure-cluster-CqcvCcdR.d.ts → failure-cluster-BLURuWG4.d.ts} +2 -3
- package/dist/failure-cluster-BLURuWG4.d.ts.map +1 -0
- package/dist/{feedback-trajectory-CacHpxsp.d.ts → feedback-trajectory-juOozjAc.d.ts} +5 -36
- package/dist/feedback-trajectory-juOozjAc.d.ts.map +1 -0
- package/dist/fuzz.d.ts +2 -2
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-CvSN3IG1.d.ts → index-B8Ui1mr1.d.ts} +2 -2
- package/dist/{index-CvSN3IG1.d.ts.map → index-B8Ui1mr1.d.ts.map} +1 -1
- package/dist/{skill-usage-CO9OLRBx.d.ts → index-BWDrSVfw.d.ts} +11 -136
- package/dist/index-BWDrSVfw.d.ts.map +1 -0
- package/dist/index-Ba3YrbAL.d.ts +1 -0
- package/dist/{index-BnEuDAK2.d.ts → index-COQYtuRF.d.ts} +3 -3
- package/dist/{index-BnEuDAK2.d.ts.map → index-COQYtuRF.d.ts.map} +1 -1
- package/dist/{index-C3ssXVLv.d.ts → index-CvjYbU0D.d.ts} +2 -2
- package/dist/{index-C3ssXVLv.d.ts.map → index-CvjYbU0D.d.ts.map} +1 -1
- package/dist/{index-DPPGNJ_R.d.ts → index-DSmEylT9.d.ts} +17 -115
- package/dist/index-DSmEylT9.d.ts.map +1 -0
- package/dist/index.d.ts +2397 -5308
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5914 -10496
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CgX_s0Ez.d.ts → insight-report-CRi-Ufrj.d.ts} +4 -4
- package/dist/{insight-report-CgX_s0Ez.d.ts.map → insight-report-CRi-Ufrj.d.ts.map} +1 -1
- package/dist/{integrity-MLzHOfV9.js → integrity-Cy9WHAtb.js} +2 -2
- package/dist/{integrity-MLzHOfV9.js.map → integrity-Cy9WHAtb.js.map} +1 -1
- package/dist/{integrity-DY6tIbl0.js → integrity-DysDBWDu.js} +2 -2
- package/dist/{integrity-DY6tIbl0.js.map → integrity-DysDBWDu.js.map} +1 -1
- package/dist/{integrity-BuqEKu-x.d.ts → integrity-OrcI9Nau.d.ts} +3 -3
- package/dist/{integrity-BuqEKu-x.d.ts.map → integrity-OrcI9Nau.d.ts.map} +1 -1
- package/dist/internal-BDHPCnjk.js +230 -0
- package/dist/internal-BDHPCnjk.js.map +1 -0
- package/dist/judge-calibration-C5CbMYce.d.ts +117 -0
- package/dist/judge-calibration-C5CbMYce.d.ts.map +1 -0
- package/dist/judge-calibration-DZkWrm5H.js +317 -0
- package/dist/judge-calibration-DZkWrm5H.js.map +1 -0
- package/dist/{kind-factory-B8-r8-y8.js → kind-factory-CPmSd58s.js} +208 -208
- package/dist/{kind-factory-B8-r8-y8.js.map → kind-factory-CPmSd58s.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-DXZIqu17.js → ledger-core-BmZt19oQ.js} +110 -110
- package/dist/{ledger-core-DXZIqu17.js.map → ledger-core-BmZt19oQ.js.map} +1 -1
- package/dist/{llm-client-DzvMUsS_.js → llm-client-d0-2TT1g.js} +4 -64
- package/dist/{llm-client-DzvMUsS_.js.map → llm-client-d0-2TT1g.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CQdVeM8k.js → llm-judge-dZ8P6nGI.js} +3349 -5401
- package/dist/llm-judge-dZ8P6nGI.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +3 -3
- package/dist/{metrics-C9YY1OcL.js → metrics-Cl0L1KUy.js} +2 -108
- package/dist/{metrics-C9YY1OcL.js.map → metrics-Cl0L1KUy.js.map} +1 -1
- package/dist/{mint-B2O60ACG.js → mint-Dj9Ww_3I.js} +4 -4
- package/dist/{mint-B2O60ACG.js.map → mint-Dj9Ww_3I.js.map} +1 -1
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts → multi-layer-verifier-DIguZc8Z.d.ts} +3 -3
- package/dist/{multi-layer-verifier-DnAqwl0h.d.ts.map → multi-layer-verifier-DIguZc8Z.d.ts.map} +1 -1
- package/dist/multiplicity-DIWHvysC.d.ts +43 -0
- package/dist/multiplicity-DIWHvysC.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +3 -3
- package/dist/multishot/index.js +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-8r6WUfHc.js → opencode-sqlite-DJWAXLms.js} +2 -2
- package/dist/{opencode-sqlite-8r6WUfHc.js.map → opencode-sqlite-DJWAXLms.js.map} +1 -1
- package/dist/package-version-D7lQHt_-.js +34 -0
- package/dist/package-version-D7lQHt_-.js.map +1 -0
- package/dist/paired-arms-D-XRF_fy.js +1045 -0
- package/dist/paired-arms-D-XRF_fy.js.map +1 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts +251 -0
- package/dist/paired-promotion-decision-CGzg0cI_.d.ts.map +1 -0
- package/dist/paired-tests-BHIhYVdu.js +213 -0
- package/dist/paired-tests-BHIhYVdu.js.map +1 -0
- package/dist/pareto-BqNW3LJR.d.ts +117 -0
- package/dist/pareto-BqNW3LJR.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +3 -64
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pipelines/index.js +4 -284
- package/dist/pipelines/index.js.map +1 -1
- package/dist/power-and-mde-CHIrXJll.js +195 -0
- package/dist/power-and-mde-CHIrXJll.js.map +1 -0
- package/dist/{promotion-policy-CrLrmys8.js → power-preflight-DEw-uC7q.js} +4 -184
- package/dist/power-preflight-DEw-uC7q.js.map +1 -0
- package/dist/pre-registration-CZwSFQS4.d.ts +577 -0
- package/dist/pre-registration-CZwSFQS4.d.ts.map +1 -0
- package/dist/{prime-protocol-BfSalTfR.js → prime-protocol-6tZTVsWm.js} +72 -21
- package/dist/prime-protocol-6tZTVsWm.js.map +1 -0
- package/dist/produced-state-DU79a81m.js +586 -0
- package/dist/produced-state-DU79a81m.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/promotion-policy-D0nPhkSy.d.ts +134 -0
- package/dist/promotion-policy-D0nPhkSy.d.ts.map +1 -0
- package/dist/promotion-policy-xzA40Evo.js +186 -0
- package/dist/promotion-policy-xzA40Evo.js.map +1 -0
- package/dist/{skillopt-optimization-method-USDKhxSA.d.ts → provenance-CCdxgLDT.d.ts} +57 -478
- package/dist/provenance-CCdxgLDT.d.ts.map +1 -0
- package/dist/registry-oJeeI4-a.d.ts +178 -0
- package/dist/registry-oJeeI4-a.d.ts.map +1 -0
- package/dist/{release-report-DA2BCu5a.d.ts → release-confidence-BFRE5WSp.d.ts} +180 -112
- package/dist/release-confidence-BFRE5WSp.d.ts.map +1 -0
- package/dist/{release-report-BUYmoKo2.js → release-confidence-CxDuiAev.js} +125 -217
- package/dist/release-confidence-CxDuiAev.js.map +1 -0
- package/dist/reporting.d.ts +6 -5
- package/dist/reporting.js +7 -5
- package/dist/{researcher-BchpD55R.d.ts → researcher-DaN4GST-.d.ts} +7 -40
- package/dist/{researcher-BchpD55R.d.ts.map → researcher-DaN4GST-.d.ts.map} +1 -1
- package/dist/{reward-hacking-BDToousL.js → reward-hacking-DNgjilrV.js} +3 -3
- package/dist/reward-hacking-DNgjilrV.js.map +1 -0
- package/dist/{reward-hacking-BI0OMAlo.d.ts → reward-hacking-DSSTuI9r.d.ts} +5 -5
- package/dist/{reward-hacking-BI0OMAlo.d.ts.map → reward-hacking-DSSTuI9r.d.ts.map} +1 -1
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +11 -10
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +4 -4
- package/dist/{rollout-DRrksrcV.js → rollout-BWtw0I_6.js} +3 -3
- package/dist/{rollout-DRrksrcV.js.map → rollout-BWtw0I_6.js.map} +1 -1
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts → rubric-predictive-validity-BgxtKe4G.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DvkjPYCe.d.ts.map → rubric-predictive-validity-BgxtKe4G.d.ts.map} +1 -1
- package/dist/{rubric-predictive-validity-BRR632r1.js → rubric-predictive-validity-Cwwyd7ah.js} +2 -2
- package/dist/{rubric-predictive-validity-BRR632r1.js.map → rubric-predictive-validity-Cwwyd7ah.js.map} +1 -1
- package/dist/{run-record-BmSPWXJR.js → run-record-BvHPVS-i.js} +3 -3
- package/dist/{run-record-BmSPWXJR.js.map → run-record-BvHPVS-i.js.map} +1 -1
- package/dist/{run-record-CF4Dwpxr.d.ts → run-record-CKiihE6f.d.ts} +4 -4
- package/dist/{run-record-CF4Dwpxr.d.ts.map → run-record-CKiihE6f.d.ts.map} +1 -1
- package/dist/{proposal-findings-2GIUo1et.js → run-score-lDzV0X8j.js} +31 -31
- package/dist/run-score-lDzV0X8j.js.map +1 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts +70 -0
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -0
- package/dist/{schema-Cef2cFmb2.d.ts → schema-Cef2cFmb.d.ts} +1 -1
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-BiJxScqe.js +406 -0
- package/dist/semantic-concept-judge-BiJxScqe.js.map +1 -0
- package/dist/{sequential-D-BLJBKU.js → sequential-C458DXNf.js} +4 -3
- package/dist/{sequential-D-BLJBKU.js.map → sequential-C458DXNf.js.map} +1 -1
- package/dist/sequential-eprocess-CbUt2htw.js +83 -0
- package/dist/sequential-eprocess-CbUt2htw.js.map +1 -0
- package/dist/series-convergence-D1cL1f-4.d.ts +129 -0
- package/dist/series-convergence-D1cL1f-4.d.ts.map +1 -0
- package/dist/{server-iu0ede49.js → server-ulsOdrTI.js} +5 -21
- package/dist/server-ulsOdrTI.js.map +1 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts +427 -0
- package/dist/skillopt-optimization-method-C5cotF4E.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js +1999 -0
- package/dist/skillopt-optimization-method-jjdnc3YK.js.map +1 -0
- package/dist/{statistical-heldout-TQ-4CYiN.d.ts → statistical-heldout-DoFF5KJX.d.ts} +5 -278
- package/dist/statistical-heldout-DoFF5KJX.d.ts.map +1 -0
- package/dist/{store-otlp-Dw8PPIlL.js → store-otlp-CsptLYpN.js} +3 -3
- package/dist/{store-otlp-Dw8PPIlL.js.map → store-otlp-CsptLYpN.js.map} +1 -1
- package/dist/store-tool-spans-Cq9mFd-q.js +667 -0
- package/dist/store-tool-spans-Cq9mFd-q.js.map +1 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts +342 -0
- package/dist/store-tool-spans-DbDLOBOb.d.ts.map +1 -0
- package/dist/student-t-CvBq2mve.js +38 -0
- package/dist/student-t-CvBq2mve.js.map +1 -0
- package/dist/{summary-report-Lf-5I7xh.js → summary-report-Blysd6Z2.js} +6 -3
- package/dist/{summary-report-Lf-5I7xh.js.map → summary-report-Blysd6Z2.js.map} +1 -1
- package/dist/{summary-report-BNR7DWTj.d.ts → summary-report-CaL-Hnxt.d.ts} +5 -5
- package/dist/{summary-report-BNR7DWTj.d.ts.map → summary-report-CaL-Hnxt.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +391 -3
- package/dist/supervisor-run/index.d.ts.map +1 -0
- package/dist/supervisor-run/index.js +1689 -2
- package/dist/{supervisor-run-D_sokXcO.js.map → supervisor-run/index.js.map} +1 -1
- package/dist/{extract-usage-CdZdoj1s.js → task-failure-attributes-CpQ4y5RD.js} +5 -157
- package/dist/task-failure-attributes-CpQ4y5RD.js.map +1 -0
- package/dist/{tool-groups-CZPGGlHf.d.ts → tool-groups-Cteb03Ps.d.ts} +3 -3
- package/dist/tool-groups-Cteb03Ps.d.ts.map +1 -0
- package/dist/tool-waste-BDdBZG1F.js +803 -0
- package/dist/tool-waste-BDdBZG1F.js.map +1 -0
- package/dist/tool-waste-DjRDEsuI.d.ts +128 -0
- package/dist/tool-waste-DjRDEsuI.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +14 -5
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +35 -7
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +406 -7
- package/dist/traces.d.ts.map +1 -0
- package/dist/traces.js +1011 -10
- package/dist/traces.js.map +1 -0
- package/dist/trajectory-replay/index.d.ts +16 -3
- package/dist/trajectory-replay/index.d.ts.map +1 -1
- package/dist/trajectory-replay/index.js +52 -5
- package/dist/trajectory-replay/index.js.map +1 -1
- package/dist/types-BEPZc6eo.d.ts +93 -0
- package/dist/types-BEPZc6eo.d.ts.map +1 -0
- package/dist/{usage-receipt-t7vAzCRQ.js → types-BI4fT3HN.js} +67 -67
- package/dist/types-BI4fT3HN.js.map +1 -0
- package/dist/{types-BnjdJ70P.d.ts → types-BZ59Ahr8.d.ts} +5 -5
- package/dist/{types-BnjdJ70P.d.ts.map → types-BZ59Ahr8.d.ts.map} +1 -1
- package/dist/{types-D216SgwM.d.ts → types-Cx3YUh2r.d.ts} +4 -241
- package/dist/types-Cx3YUh2r.d.ts.map +1 -0
- package/dist/{types-BvDKaULh.d.ts → types-DLQx4mKU.d.ts} +4 -4
- package/dist/{types-BvDKaULh.d.ts.map → types-DLQx4mKU.d.ts.map} +1 -1
- package/dist/{types-D4mog56g.d.ts → types-yLK8gXE9.d.ts} +2 -2
- package/dist/{types-D4mog56g.d.ts.map → types-yLK8gXE9.d.ts.map} +1 -1
- package/dist/verdict-BndeTAh_.js +61 -0
- package/dist/verdict-BndeTAh_.js.map +1 -0
- package/dist/verdict-E4eRNf7-.d.ts +392 -0
- package/dist/verdict-E4eRNf7-.d.ts.map +1 -0
- package/dist/{verdict-cache-BCcOh0kF.js → verdict-cache-mZf5FEiY.js} +3 -55
- package/dist/{verdict-cache-BCcOh0kF.js.map → verdict-cache-mZf5FEiY.js.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +3 -3
- package/docs/control-runtime.md +3 -42
- package/docs/experiment.md +0 -1
- package/docs/feature-guide.md +2 -2
- package/docs/trace-repair-grader.md +1 -0
- package/docs/trajectory-replay.md +1 -0
- package/docs/verdicts.md +43 -0
- package/docs/verification-strategies.md +3 -2
- package/package.json +6 -11
- package/dist/agent-profile-DPi7IZg7.d.ts.map +0 -1
- package/dist/analyze-runs-C30yljDJ.js.map +0 -1
- package/dist/baseline-CavEbRyH.d.ts +0 -136
- package/dist/baseline-CavEbRyH.d.ts.map +0 -1
- package/dist/benchmark-command-BteMFN62.js.map +0 -1
- package/dist/benchmarks-Dzs8CKb1.js +0 -755
- package/dist/benchmarks-Dzs8CKb1.js.map +0 -1
- package/dist/campaign-C2TTzQII.js.map +0 -1
- package/dist/completion-verifier-DJA5BhPb.d.ts +0 -414
- package/dist/completion-verifier-DJA5BhPb.d.ts.map +0 -1
- package/dist/control.d.ts +0 -3
- package/dist/control.js +0 -2
- package/dist/dataset-C8xaLXdY.d.ts.map +0 -1
- package/dist/default-registry-Bf8Woqmq.d.ts.map +0 -1
- package/dist/default-registry-BmktKy8r.js.map +0 -1
- package/dist/engine-3hL-XqwJ.d.ts.map +0 -1
- package/dist/experiment-tracker-CnRICnMl.js +0 -500
- package/dist/experiment-tracker-CnRICnMl.js.map +0 -1
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +0 -1
- package/dist/extract-usage-CdZdoj1s.js.map +0 -1
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +0 -1
- package/dist/feedback-trajectory-CacHpxsp.d.ts.map +0 -1
- package/dist/index-BZ3-y4YL.d.ts +0 -391
- package/dist/index-BZ3-y4YL.d.ts.map +0 -1
- package/dist/index-CQTZ-4XN.d.ts.map +0 -1
- package/dist/index-DPPGNJ_R.d.ts.map +0 -1
- package/dist/index-YE4KdKbO2.d.ts +0 -335
- package/dist/index-YE4KdKbO2.d.ts.map +0 -1
- package/dist/paired-arms-iZ08VFMN.js +0 -260
- package/dist/paired-arms-iZ08VFMN.js.map +0 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +0 -114
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +0 -1
- package/dist/prime-protocol-BfSalTfR.js.map +0 -1
- package/dist/promotion-policy-Ckjhzg_4.d.ts +0 -289
- package/dist/promotion-policy-Ckjhzg_4.d.ts.map +0 -1
- package/dist/promotion-policy-CrLrmys8.js.map +0 -1
- package/dist/proposal-findings-2GIUo1et.js.map +0 -1
- package/dist/propose-review-control-dSNPjFUH.js +0 -1458
- package/dist/propose-review-control-dSNPjFUH.js.map +0 -1
- package/dist/release-report-BUYmoKo2.js.map +0 -1
- package/dist/release-report-DA2BCu5a.d.ts.map +0 -1
- package/dist/replay-CohS93nE.js +0 -1859
- package/dist/replay-CohS93nE.js.map +0 -1
- package/dist/replay-DbhZ4Ked.d.ts +0 -834
- package/dist/replay-DbhZ4Ked.d.ts.map +0 -1
- package/dist/reward-hacking-BDToousL.js.map +0 -1
- package/dist/run-evidence-Dhi3C81V.d.ts +0 -225
- package/dist/run-evidence-Dhi3C81V.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +0 -1
- package/dist/semantic-concept-judge-D1z-KepS.js +0 -767
- package/dist/semantic-concept-judge-D1z-KepS.js.map +0 -1
- package/dist/series-convergence-ofsqPWhs.d.ts +0 -35
- package/dist/series-convergence-ofsqPWhs.d.ts.map +0 -1
- package/dist/server-iu0ede49.js.map +0 -1
- package/dist/single-run-lock-DFWHEB09.js.map +0 -1
- package/dist/skill-usage-CO9OLRBx.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CQdVeM8k.js.map +0 -1
- package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +0 -1
- package/dist/statistical-heldout-TQ-4CYiN.d.ts.map +0 -1
- package/dist/statistics-ByxzSiOM.js +0 -2212
- package/dist/statistics-ByxzSiOM.js.map +0 -1
- package/dist/statistics-D6Uebe_4.d.ts +0 -968
- package/dist/statistics-D6Uebe_4.d.ts.map +0 -1
- package/dist/supervisor-run-D_sokXcO.js +0 -1690
- package/dist/test-graded-scenario-D1TaI2va.d.ts +0 -141
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +0 -1
- package/dist/test-graded-scenario-JHcKQNpq.js +0 -318
- package/dist/test-graded-scenario-JHcKQNpq.js.map +0 -1
- package/dist/tool-groups-CZPGGlHf.d.ts.map +0 -1
- package/dist/tool-use-metrics-DEGMKycK.js +0 -370
- package/dist/tool-use-metrics-DEGMKycK.js.map +0 -1
- package/dist/types-D216SgwM.d.ts.map +0 -1
- package/dist/usage-receipt-t7vAzCRQ.js.map +0 -1
- package/dist/verdict-DExhxfgR.d.ts +0 -201
- package/dist/verdict-DExhxfgR.d.ts.map +0 -1
|
@@ -1,639 +1,26 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { t as
|
|
1
|
+
import { s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
|
|
2
|
+
import { c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, s as agentProfileHash, t as extractProducedState } from "./produced-state-DU79a81m.js";
|
|
3
3
|
import { buildAgentProfileCell } from "./profile-cell.js";
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
6
|
-
import { r as
|
|
7
|
-
import {
|
|
8
|
-
import {
|
|
9
|
-
import {
|
|
10
|
-
import {
|
|
11
|
-
import {
|
|
12
|
-
import {
|
|
13
|
-
import {
|
|
14
|
-
import {
|
|
15
|
-
import { a as scoreAnalystFindings } from "./benchmark-
|
|
16
|
-
import
|
|
17
|
-
import
|
|
4
|
+
import { t as comparePairedArms } from "./paired-arms-D-XRF_fy.js";
|
|
5
|
+
import { r as pairedBootstrap } from "./paired-tests-BHIhYVdu.js";
|
|
6
|
+
import { o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord } from "./run-record-BvHPVS-i.js";
|
|
7
|
+
import { n as contentHash } from "./verdict-cache-mZf5FEiY.js";
|
|
8
|
+
import { K as runCampaign, Q as summarizeBackendIntegrity, Z as assertRealBackend, h as labelTrustRank, k as surfaceContentHash, q as planCampaignRun, w as assertCodeSurfaceIdentity } from "./llm-judge-dZ8P6nGI.js";
|
|
9
|
+
import { l as campaignCellToRunRecord } from "./reward-hacking-DNgjilrV.js";
|
|
10
|
+
import { t as CostAccountingIncompleteError } from "./cost-ledger-BSe92yAV.js";
|
|
11
|
+
import { t as FileLedgerJournal, u as mapConcurrent } from "./ledger-core-BmZt19oQ.js";
|
|
12
|
+
import { r as canonicalString } from "./canonical-D-XsTQ6_.js";
|
|
13
|
+
import { C as SEARCH_LEDGER_FILE_CONTEXT, E as SearchLedgerIntegrityError, T as SearchLedgerError, w as SearchLedgerConflictError } from "./external-optimizer-subprocess-DrJ9hR8u.js";
|
|
14
|
+
import { o as pairHoldout } from "./power-preflight-DEw-uC7q.js";
|
|
15
|
+
import { a as scoreAnalystFindings } from "./benchmark-BhT16ep9.js";
|
|
16
|
+
import "./external-optimizer-process-BTiNB-RH.js";
|
|
17
|
+
import "./skillopt-optimization-method-jjdnc3YK.js";
|
|
18
|
+
import { createHash } from "node:crypto";
|
|
18
19
|
import { closeSync, constants, existsSync, fstatSync, lstatSync, mkdirSync, mkdtempSync, openSync, readFileSync, readSync, readdirSync, readlinkSync, realpathSync, rmSync, statSync, writeFileSync } from "node:fs";
|
|
20
|
+
import { z } from "zod";
|
|
19
21
|
import { basename, dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
|
|
20
|
-
import { createHash, randomUUID } from "node:crypto";
|
|
21
|
-
import { harnessSupportsModel } from "@tangle-network/agent-interface";
|
|
22
|
-
import { execFileSync } from "node:child_process";
|
|
23
22
|
import { devNull, tmpdir } from "node:os";
|
|
24
|
-
|
|
25
|
-
/**
|
|
26
|
-
* Completion verifier — the task-completion oracle.
|
|
27
|
-
*
|
|
28
|
-
* Answers the only eval question that is not a proxy: did the agent actually
|
|
29
|
-
* COMPLETE the task — produce every required deliverable, persisted and
|
|
30
|
-
* correct — rather than describe what should be done. A fluent transcript
|
|
31
|
-
* that never produces the artifact scores zero here.
|
|
32
|
-
*
|
|
33
|
-
* Per requirement, a two-stage check:
|
|
34
|
-
* 1. Structural — a produced item (vault artifact / approved proposal /
|
|
35
|
-
* tool call) of the right kind is matched against the requirement and
|
|
36
|
-
* carries non-empty content. Deterministic; no LLM.
|
|
37
|
-
* 2. Correctness — only if structurally present AND the matched item
|
|
38
|
-
* carries content, one targeted check decides whether that item
|
|
39
|
-
* actually fulfils the requirement. A hallucinated artifact fails here;
|
|
40
|
-
* an absent one already failed stage 1.
|
|
41
|
-
*
|
|
42
|
-
* `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —
|
|
43
|
-
* checker failures — are excluded from the denominator, never scored as
|
|
44
|
-
* zeros). Quality dimensions are meaningless on an incomplete task — callers
|
|
45
|
-
* gate on `fullyComplete` / `completionRate` before scoring quality.
|
|
46
|
-
*/
|
|
47
|
-
/**
|
|
48
|
-
* Construct a `CompletionVerdict` from the per-requirement checks, deriving
|
|
49
|
-
* `completionRate` / `fullyComplete` and the spine fields (`valid` =
|
|
50
|
-
* `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
|
|
51
|
-
* requirements — a verdict over nothing is a misconfiguration, mirroring
|
|
52
|
-
* `verifyCompletion`'s gold-spec guard.
|
|
53
|
-
*/
|
|
54
|
-
function completionVerdict(input) {
|
|
55
|
-
if (input.requirements.length === 0) throw new Error(`completionVerdict: task '${input.taskId}' has no requirement checks — nothing to derive a verdict from`);
|
|
56
|
-
const measurable = input.requirements.filter((r) => !r.unmeasured);
|
|
57
|
-
const unmeasuredCount = input.requirements.length - measurable.length;
|
|
58
|
-
if (measurable.length === 0) throw new Error(`completionVerdict: task '${input.taskId}' has no measurable requirements — all ${input.requirements.length} correctness checks failed (${input.requirements[0]?.unmeasuredReason ?? "unknown reason"})`);
|
|
59
|
-
const satisfiedCount = measurable.filter((r) => r.satisfied).length;
|
|
60
|
-
const completionRate = satisfiedCount / measurable.length;
|
|
61
|
-
const fullyComplete = unmeasuredCount === 0 && satisfiedCount === measurable.length;
|
|
62
|
-
return {
|
|
63
|
-
taskId: input.taskId,
|
|
64
|
-
requirements: input.requirements,
|
|
65
|
-
completionRate,
|
|
66
|
-
fullyComplete,
|
|
67
|
-
unmeasuredCount,
|
|
68
|
-
valid: fullyComplete,
|
|
69
|
-
score: completionRate
|
|
70
|
-
};
|
|
71
|
-
}
|
|
72
|
-
const STOPWORDS = /* @__PURE__ */ new Set([
|
|
73
|
-
"the",
|
|
74
|
-
"a",
|
|
75
|
-
"an",
|
|
76
|
-
"of",
|
|
77
|
-
"for",
|
|
78
|
-
"and",
|
|
79
|
-
"or",
|
|
80
|
-
"to",
|
|
81
|
-
"in",
|
|
82
|
-
"on",
|
|
83
|
-
"with",
|
|
84
|
-
"by"
|
|
85
|
-
]);
|
|
86
|
-
const REQUIREMENT_FORM_STOPWORDS = /* @__PURE__ */ new Set([
|
|
87
|
-
"generated",
|
|
88
|
-
"generate",
|
|
89
|
-
"view",
|
|
90
|
-
"render",
|
|
91
|
-
"rendered",
|
|
92
|
-
"persisted",
|
|
93
|
-
"persist",
|
|
94
|
-
"artifact",
|
|
95
|
-
"file",
|
|
96
|
-
"document",
|
|
97
|
-
"note",
|
|
98
|
-
"proposal",
|
|
99
|
-
"deliverable",
|
|
100
|
-
"output",
|
|
101
|
-
"created",
|
|
102
|
-
"create",
|
|
103
|
-
"produce",
|
|
104
|
-
"produced",
|
|
105
|
-
"flag"
|
|
106
|
-
]);
|
|
107
|
-
const MATCH_THRESHOLD = .5;
|
|
108
|
-
const MIN_CONTENT_CHARS = 50;
|
|
109
|
-
function tokens(s, extraStop) {
|
|
110
|
-
return new Set(s.toLowerCase().split(/[^a-z0-9]+/).filter((t) => t.length > 1 && !STOPWORDS.has(t) && !extraStop?.has(t)));
|
|
111
|
-
}
|
|
112
|
-
/**
|
|
113
|
-
* Recall of the requirement's tokens within a candidate's identifying text.
|
|
114
|
-
* Recall, not Jaccard — a candidate's path/id legitimately carries extra
|
|
115
|
-
* tokens the requirement does not name. The requirement side drops
|
|
116
|
-
* deliverable-FORM vocabulary so recall keys on the distinctive domain tokens.
|
|
117
|
-
*/
|
|
118
|
-
function tokenRecall(requirementText, candidateText) {
|
|
119
|
-
const req = tokens(requirementText, REQUIREMENT_FORM_STOPWORDS);
|
|
120
|
-
if (req.size === 0) return 0;
|
|
121
|
-
const cand = tokens(candidateText);
|
|
122
|
-
let hit = 0;
|
|
123
|
-
for (const t of req) if (cand.has(t)) hit++;
|
|
124
|
-
return hit / req.size;
|
|
125
|
-
}
|
|
126
|
-
function artifactCandidates(req, reqIndex, artifacts) {
|
|
127
|
-
const reqText = `${req.title} ${req.category ?? ""}`;
|
|
128
|
-
const out = [];
|
|
129
|
-
artifacts.forEach((a, i) => {
|
|
130
|
-
if ((a.content ?? "").trim().length < MIN_CONTENT_CHARS) return;
|
|
131
|
-
let score = tokenRecall(reqText, `${a.path ?? ""} ${a.kind} ${(a.content ?? "").slice(0, 4e3)}`);
|
|
132
|
-
if (req.category && a.kind && req.category.toLowerCase() === a.kind.toLowerCase()) score = Math.max(score, 1);
|
|
133
|
-
if (score < MATCH_THRESHOLD) return;
|
|
134
|
-
out.push({
|
|
135
|
-
reqIndex,
|
|
136
|
-
itemKey: `artifact:${i}`,
|
|
137
|
-
score,
|
|
138
|
-
evidence: `artifact '${a.path ?? a.kind}' matched (token recall ${score.toFixed(2)})`,
|
|
139
|
-
content: a.content ?? null
|
|
140
|
-
});
|
|
141
|
-
});
|
|
142
|
-
return out;
|
|
143
|
-
}
|
|
144
|
-
function proposalCandidates(req, reqIndex, proposals) {
|
|
145
|
-
const reqText = `${req.title} ${req.category ?? ""}`;
|
|
146
|
-
const out = [];
|
|
147
|
-
for (const p of proposals) {
|
|
148
|
-
if (p.status !== "approved") continue;
|
|
149
|
-
const body = (p.content ?? "").trim();
|
|
150
|
-
if (body.length < MIN_CONTENT_CHARS) continue;
|
|
151
|
-
const score = tokenRecall(reqText, `${p.title} ${body}`);
|
|
152
|
-
if (score < MATCH_THRESHOLD) continue;
|
|
153
|
-
out.push({
|
|
154
|
-
reqIndex,
|
|
155
|
-
itemKey: `proposal:${p.id}`,
|
|
156
|
-
score,
|
|
157
|
-
evidence: `approved proposal '${p.title}' matched (token recall ${score.toFixed(2)})`,
|
|
158
|
-
content: body
|
|
159
|
-
});
|
|
160
|
-
}
|
|
161
|
-
return out;
|
|
162
|
-
}
|
|
163
|
-
function toolCallCandidates(req, reqIndex, toolCalls) {
|
|
164
|
-
const out = [];
|
|
165
|
-
toolCalls.forEach((name, i) => {
|
|
166
|
-
const score = tokenRecall(req.title, name);
|
|
167
|
-
if (score < MATCH_THRESHOLD) return;
|
|
168
|
-
out.push({
|
|
169
|
-
reqIndex,
|
|
170
|
-
itemKey: `tool:${i}`,
|
|
171
|
-
score,
|
|
172
|
-
evidence: `tool call '${name}' matched (token recall ${score.toFixed(2)})`,
|
|
173
|
-
content: null
|
|
174
|
-
});
|
|
175
|
-
});
|
|
176
|
-
return out;
|
|
177
|
-
}
|
|
178
|
-
/**
|
|
179
|
-
* Verify whether a run completed the task. `checkCorrectness` is injected —
|
|
180
|
-
* `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
|
|
181
|
-
*
|
|
182
|
-
* Throws on a gold spec with no requirements: an eval task that requires
|
|
183
|
-
* nothing is a misconfiguration, not a vacuously-complete task.
|
|
184
|
-
*/
|
|
185
|
-
async function verifyCompletion(gold, state, checkCorrectness) {
|
|
186
|
-
if (gold.requirements.length === 0) throw new Error(`verifyCompletion: task '${gold.taskId}' has no requirements — malformed gold spec`);
|
|
187
|
-
const candidates = [];
|
|
188
|
-
gold.requirements.forEach((req, i) => {
|
|
189
|
-
const by = req.satisfiedBy ?? "any";
|
|
190
|
-
if (by === "artifact" || by === "any") candidates.push(...artifactCandidates(req, i, state.artifacts));
|
|
191
|
-
if (by === "proposal" || by === "any") candidates.push(...proposalCandidates(req, i, state.proposals));
|
|
192
|
-
if (by === "tool-call" || by === "any") candidates.push(...toolCallCandidates(req, i, state.toolCalls));
|
|
193
|
-
});
|
|
194
|
-
candidates.sort((a, b) => b.score - a.score);
|
|
195
|
-
const assigned = /* @__PURE__ */ new Map();
|
|
196
|
-
const itemTaken = /* @__PURE__ */ new Set();
|
|
197
|
-
for (const c of candidates) {
|
|
198
|
-
if (assigned.has(c.reqIndex) || itemTaken.has(c.itemKey)) continue;
|
|
199
|
-
assigned.set(c.reqIndex, c);
|
|
200
|
-
itemTaken.add(c.itemKey);
|
|
201
|
-
}
|
|
202
|
-
const requirements = [];
|
|
203
|
-
for (let i = 0; i < gold.requirements.length; i++) {
|
|
204
|
-
const req = gold.requirements[i];
|
|
205
|
-
const match = assigned.get(i);
|
|
206
|
-
const evidence = [];
|
|
207
|
-
let correct = null;
|
|
208
|
-
let unmeasuredReason;
|
|
209
|
-
if (match) {
|
|
210
|
-
evidence.push(match.evidence);
|
|
211
|
-
if (match.content !== null) try {
|
|
212
|
-
const r = await checkCorrectness(req, match.content);
|
|
213
|
-
correct = r.correct;
|
|
214
|
-
evidence.push(`correctness: ${r.correct ? "pass" : "fail"} — ${r.reason}`);
|
|
215
|
-
} catch (err) {
|
|
216
|
-
unmeasuredReason = err instanceof JudgeParseError ? `checker response unparseable after retry: ${err.raw.slice(0, 200)}` : `checker call failed: ${err instanceof Error ? err.message : String(err)}`;
|
|
217
|
-
evidence.push(`correctness: UNMEASURED — ${unmeasuredReason}`);
|
|
218
|
-
}
|
|
219
|
-
else evidence.push("correctness: not assessed — matched item carries no content");
|
|
220
|
-
} else {
|
|
221
|
-
const by = req.satisfiedBy ?? "any";
|
|
222
|
-
const kind = by === "any" ? "artifact/proposal/tool-call" : by;
|
|
223
|
-
evidence.push(`no produced ${kind} matched this requirement`);
|
|
224
|
-
}
|
|
225
|
-
const structurallyPresent = match !== void 0;
|
|
226
|
-
const unmeasured = unmeasuredReason !== void 0;
|
|
227
|
-
const satisfied = structurallyPresent && !unmeasured && correct !== false;
|
|
228
|
-
requirements.push({
|
|
229
|
-
reqId: req.reqId,
|
|
230
|
-
title: req.title,
|
|
231
|
-
structurallyPresent,
|
|
232
|
-
correct,
|
|
233
|
-
satisfied,
|
|
234
|
-
...unmeasured ? {
|
|
235
|
-
unmeasured: true,
|
|
236
|
-
unmeasuredReason
|
|
237
|
-
} : {},
|
|
238
|
-
evidence
|
|
239
|
-
});
|
|
240
|
-
}
|
|
241
|
-
return completionVerdict({
|
|
242
|
-
taskId: gold.taskId,
|
|
243
|
-
requirements
|
|
244
|
-
});
|
|
245
|
-
}
|
|
246
|
-
/**
|
|
247
|
-
* Parse the correctness checker's model response. Tolerates a response
|
|
248
|
-
* truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the
|
|
249
|
-
* verdict boolean usually lands in the first few tokens, so a recovered
|
|
250
|
-
* prefix with a boolean `correct` is a real measurement, not a guess.
|
|
251
|
-
* Fails loud (JudgeParseError) when no boolean verdict is recoverable.
|
|
252
|
-
*/
|
|
253
|
-
function parseCorrectnessResponse(raw) {
|
|
254
|
-
const readVerdict = (candidate) => {
|
|
255
|
-
if (candidate === null || typeof candidate !== "object") return null;
|
|
256
|
-
const { correct, reason } = candidate;
|
|
257
|
-
if (typeof correct !== "boolean") return null;
|
|
258
|
-
return {
|
|
259
|
-
correct,
|
|
260
|
-
reason: typeof reason === "string" ? reason : ""
|
|
261
|
-
};
|
|
262
|
-
};
|
|
263
|
-
const match = raw.match(/\{[\s\S]*\}/);
|
|
264
|
-
if (match) try {
|
|
265
|
-
const strict = readVerdict(JSON.parse(match[0]));
|
|
266
|
-
if (strict) return strict;
|
|
267
|
-
} catch {}
|
|
268
|
-
const start = raw.indexOf("{");
|
|
269
|
-
if (start !== -1) {
|
|
270
|
-
const recovered = readVerdict(recoverTruncatedJson(raw.slice(start)));
|
|
271
|
-
if (recovered) return recovered;
|
|
272
|
-
}
|
|
273
|
-
throw new JudgeParseError("correctness-checker", raw);
|
|
274
|
-
}
|
|
275
|
-
/**
|
|
276
|
-
* Production `CorrectnessChecker` — one LLM call per matched artifact,
|
|
277
|
-
* deterministic (temperature 0), structured JSON out. Judges fulfilment
|
|
278
|
-
* only: a plan, a gesture, or a description of what should be done does not
|
|
279
|
-
* fulfil a requirement — the artifact must BE the deliverable.
|
|
280
|
-
*/
|
|
281
|
-
function createLlmCorrectnessChecker(chat, opts = {}) {
|
|
282
|
-
const model = opts.model ?? "claude-sonnet-4-6";
|
|
283
|
-
const maxContentChars = opts.maxContentChars ?? 8e3;
|
|
284
|
-
const maxAttempts = opts.maxAttempts ?? 2;
|
|
285
|
-
const costLedger = opts.costLedger ?? new CostLedger();
|
|
286
|
-
const sink = opts.rawSink;
|
|
287
|
-
const record = async (event) => {
|
|
288
|
-
try {
|
|
289
|
-
await sink?.record(event);
|
|
290
|
-
} catch {}
|
|
291
|
-
};
|
|
292
|
-
return async (requirement, content) => {
|
|
293
|
-
const request = {
|
|
294
|
-
model,
|
|
295
|
-
messages: [{
|
|
296
|
-
role: "system",
|
|
297
|
-
content: "You verify whether a produced work artifact actually fulfils a stated requirement. Judge fulfilment only — is the deliverable substantively present and on-point — not polish. A plan to do it later, a vague gesture, or a description of what should be done does NOT fulfil a requirement; the artifact must BE the deliverable. Respond with a single JSON object: {\"correct\": boolean, \"reason\": string (<= 30 words)}."
|
|
298
|
-
}, {
|
|
299
|
-
role: "user",
|
|
300
|
-
content: `Requirement: ${requirement.title}\n${requirement.category ? `Category: ${requirement.category}\n` : ""}\nProduced artifact:\n${content.slice(0, maxContentChars)}`
|
|
301
|
-
}],
|
|
302
|
-
temperature: 0,
|
|
303
|
-
maxTokens: 200
|
|
304
|
-
};
|
|
305
|
-
let lastErr;
|
|
306
|
-
for (let attempt = 0; attempt < maxAttempts; attempt++) {
|
|
307
|
-
const started = Date.now();
|
|
308
|
-
await record({
|
|
309
|
-
eventId: randomUUID(),
|
|
310
|
-
provider: chat.transport,
|
|
311
|
-
model,
|
|
312
|
-
endpoint: "/chat",
|
|
313
|
-
baseUrl: "",
|
|
314
|
-
attemptIndex: attempt,
|
|
315
|
-
direction: "request",
|
|
316
|
-
timestamp: started,
|
|
317
|
-
requestBody: request,
|
|
318
|
-
redactedFields: []
|
|
319
|
-
});
|
|
320
|
-
try {
|
|
321
|
-
const paid = await costLedger.runPaidCall({
|
|
322
|
-
channel: "verifier",
|
|
323
|
-
phase: opts.costPhase ?? "completion.correctness",
|
|
324
|
-
actor: "correctness-checker",
|
|
325
|
-
model,
|
|
326
|
-
maximumCharge: chat.maximumAttempts === void 0 ? void 0 : maximumChargeForLlmRequest(request, { maximumAttempts: chat.maximumAttempts }),
|
|
327
|
-
tags: {
|
|
328
|
-
...opts.costTags,
|
|
329
|
-
requirementId: requirement.reqId,
|
|
330
|
-
attempt: String(attempt)
|
|
331
|
-
},
|
|
332
|
-
signal: opts.signal,
|
|
333
|
-
execute: (signal, callId) => chat.chat(request, {
|
|
334
|
-
signal,
|
|
335
|
-
idempotencyKey: callId
|
|
336
|
-
}),
|
|
337
|
-
receipt: costReceiptFromLlm,
|
|
338
|
-
receiptFromError: costReceiptFromLlmError
|
|
339
|
-
});
|
|
340
|
-
if (!paid.succeeded) throw paid.error;
|
|
341
|
-
const resp = paid.value;
|
|
342
|
-
assertServedModel(model, resp.servedModel, {
|
|
343
|
-
allowUnreported: true,
|
|
344
|
-
context: `correctness checker for requirement ${requirement.reqId}`
|
|
345
|
-
});
|
|
346
|
-
const raw = resp.content;
|
|
347
|
-
await record({
|
|
348
|
-
eventId: randomUUID(),
|
|
349
|
-
provider: chat.transport,
|
|
350
|
-
model,
|
|
351
|
-
endpoint: "/chat",
|
|
352
|
-
baseUrl: "",
|
|
353
|
-
attemptIndex: attempt,
|
|
354
|
-
direction: "response",
|
|
355
|
-
timestamp: Date.now(),
|
|
356
|
-
durationMs: Date.now() - started,
|
|
357
|
-
responseBody: resp,
|
|
358
|
-
redactedFields: []
|
|
359
|
-
});
|
|
360
|
-
return parseCorrectnessResponse(raw);
|
|
361
|
-
} catch (err) {
|
|
362
|
-
lastErr = err;
|
|
363
|
-
await record({
|
|
364
|
-
eventId: randomUUID(),
|
|
365
|
-
provider: chat.transport,
|
|
366
|
-
model,
|
|
367
|
-
endpoint: "/chat",
|
|
368
|
-
baseUrl: "",
|
|
369
|
-
attemptIndex: attempt,
|
|
370
|
-
direction: "error",
|
|
371
|
-
timestamp: Date.now(),
|
|
372
|
-
durationMs: Date.now() - started,
|
|
373
|
-
errorMessage: err instanceof Error ? err.message : String(err),
|
|
374
|
-
redactedFields: []
|
|
375
|
-
});
|
|
376
|
-
if (err instanceof ModelSubstitutionError) throw err;
|
|
377
|
-
}
|
|
378
|
-
}
|
|
379
|
-
throw lastErr instanceof Error ? lastErr : new Error(String(lastErr));
|
|
380
|
-
};
|
|
381
|
-
}
|
|
382
|
-
/** Stopwords for requirement-title tokenization — drops the imperative verbs
|
|
383
|
-
* ('review', 'update', …) common to deliverable titles so recall keys on the
|
|
384
|
-
* substantive nouns, not the boilerplate ask. */
|
|
385
|
-
const TITLE_STOPWORDS = /* @__PURE__ */ new Set([
|
|
386
|
-
"the",
|
|
387
|
-
"a",
|
|
388
|
-
"an",
|
|
389
|
-
"and",
|
|
390
|
-
"or",
|
|
391
|
-
"for",
|
|
392
|
-
"to",
|
|
393
|
-
"of",
|
|
394
|
-
"in",
|
|
395
|
-
"on",
|
|
396
|
-
"with",
|
|
397
|
-
"review",
|
|
398
|
-
"update",
|
|
399
|
-
"new",
|
|
400
|
-
"proposed"
|
|
401
|
-
]);
|
|
402
|
-
/**
|
|
403
|
-
* Deterministic `CorrectnessChecker` — the no-LLM counterpart to
|
|
404
|
-
* `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
|
|
405
|
-
* content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`
|
|
406
|
-
* of the requirement title's significant tokens. No network.
|
|
407
|
-
*
|
|
408
|
-
* Polarity-blind: token recall credits a negation that contains the
|
|
409
|
-
* requirement's tokens ("I will NOT produce the comparison" recalls every token
|
|
410
|
-
* of "produce the comparison"). The structural match stage is ALSO lexical, so
|
|
411
|
-
* pairing the two collapses to a single gameable gate. Use this only as an
|
|
412
|
-
* opt-in structural pre-filter or for tasks whose requirements have no polarity
|
|
413
|
-
* to invert; for produced-state grading the correctness checker MUST be semantic
|
|
414
|
-
* (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.
|
|
415
|
-
*/
|
|
416
|
-
function createTokenRecallChecker(opts = {}) {
|
|
417
|
-
const minRecall = opts.minRecall ?? .5;
|
|
418
|
-
const minLen = opts.minContentLength ?? 120;
|
|
419
|
-
return async (requirement, content) => {
|
|
420
|
-
const body = content.trim();
|
|
421
|
-
if (body.length < minLen) return {
|
|
422
|
-
correct: false,
|
|
423
|
-
reason: `content too thin (${body.length} chars) to be the deliverable`
|
|
424
|
-
};
|
|
425
|
-
const titleTokens = requirement.title.toLowerCase().split(/[^a-z0-9]+/).filter((t) => t.length > 2 && !TITLE_STOPWORDS.has(t));
|
|
426
|
-
if (titleTokens.length === 0) return {
|
|
427
|
-
correct: true,
|
|
428
|
-
reason: "requirement title has no significant tokens — structural match accepted"
|
|
429
|
-
};
|
|
430
|
-
const lower = body.toLowerCase();
|
|
431
|
-
const hits = titleTokens.filter((t) => lower.includes(t)).length;
|
|
432
|
-
return hits / titleTokens.length >= minRecall ? {
|
|
433
|
-
correct: true,
|
|
434
|
-
reason: `content recalls ${hits}/${titleTokens.length} requirement tokens`
|
|
435
|
-
} : {
|
|
436
|
-
correct: false,
|
|
437
|
-
reason: `content recalls only ${hits}/${titleTokens.length} requirement tokens`
|
|
438
|
-
};
|
|
439
|
-
};
|
|
440
|
-
}
|
|
441
|
-
//#endregion
|
|
442
|
-
//#region src/produced-state.ts
|
|
443
|
-
function artifactKind(mimeType) {
|
|
444
|
-
if (!mimeType) return "file";
|
|
445
|
-
if (mimeType.includes("json")) return "json";
|
|
446
|
-
if (mimeType.startsWith("text/")) return "text";
|
|
447
|
-
return "file";
|
|
448
|
-
}
|
|
449
|
-
/**
|
|
450
|
-
* Normalize a run's runtime event stream into `ProducedState`.
|
|
451
|
-
*
|
|
452
|
-
* Pure and total — unrecognized event types are skipped. `toolCalls` is
|
|
453
|
-
* deduplicated by name in first-seen order (completion cares about a tool's
|
|
454
|
-
* presence, not its call count). An artifact with neither a name nor a uri
|
|
455
|
-
* still yields an entry keyed by its `artifactId` so it is never silently
|
|
456
|
-
* dropped; an artifact with no `content` yields empty content, which the
|
|
457
|
-
* completion oracle's structural check then rejects on its own.
|
|
458
|
-
*/
|
|
459
|
-
function extractProducedState(events) {
|
|
460
|
-
const artifacts = [];
|
|
461
|
-
const proposals = [];
|
|
462
|
-
const toolCalls = [];
|
|
463
|
-
const seenTools = /* @__PURE__ */ new Set();
|
|
464
|
-
for (const ev of events) if (ev.type === "tool_call") {
|
|
465
|
-
const name = ev.toolName;
|
|
466
|
-
if (name && !seenTools.has(name)) {
|
|
467
|
-
seenTools.add(name);
|
|
468
|
-
toolCalls.push(name);
|
|
469
|
-
}
|
|
470
|
-
} else if (ev.type === "artifact") {
|
|
471
|
-
const a = ev;
|
|
472
|
-
artifacts.push({
|
|
473
|
-
kind: artifactKind(a.mimeType),
|
|
474
|
-
path: a.name ?? a.uri ?? a.artifactId,
|
|
475
|
-
content: a.content ?? ""
|
|
476
|
-
});
|
|
477
|
-
} else if (ev.type === "proposal_created") {
|
|
478
|
-
const p = ev;
|
|
479
|
-
proposals.push({
|
|
480
|
-
id: p.proposalId,
|
|
481
|
-
title: p.title,
|
|
482
|
-
status: p.status ?? "pending",
|
|
483
|
-
...p.content !== void 0 ? { content: p.content } : {}
|
|
484
|
-
});
|
|
485
|
-
}
|
|
486
|
-
return {
|
|
487
|
-
artifacts,
|
|
488
|
-
proposals,
|
|
489
|
-
toolCalls
|
|
490
|
-
};
|
|
491
|
-
}
|
|
492
|
-
//#endregion
|
|
493
|
-
//#region src/agent-profile.ts
|
|
494
|
-
/**
|
|
495
|
-
* The agentic coding harnesses an eval sweeps by default — the ones we care about
|
|
496
|
-
* ranking. This is the SINGLE source of that list; consumers import it instead of
|
|
497
|
-
* re-declaring their own (a re-declared list is how the fleet drifts). Pass an
|
|
498
|
-
* explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
|
|
499
|
-
* harness) to widen beyond these.
|
|
500
|
-
*/
|
|
501
|
-
const CODING_HARNESSES = [
|
|
502
|
-
"opencode",
|
|
503
|
-
"claude-code",
|
|
504
|
-
"codex",
|
|
505
|
-
"kimi-code"
|
|
506
|
-
];
|
|
507
|
-
/** Model sentinel for a vendor-locked harness that supports none of the swept models:
|
|
508
|
-
* it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
|
|
509
|
-
* resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
|
|
510
|
-
* model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
|
|
511
|
-
* table that would rot as router catalogs change. */
|
|
512
|
-
const HARNESS_NATIVE_MODEL = "default";
|
|
513
|
-
/**
|
|
514
|
-
* Expand a base profile across the harness × model matrix into the `AgentProfile[]`
|
|
515
|
-
* that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
|
|
516
|
-
* which models do we evaluate" lives, so no product hand-rolls its own harness list
|
|
517
|
-
* or column→profile mapping (the pattern that let those copies drift and silently
|
|
518
|
-
* break the harness pivot).
|
|
519
|
-
*
|
|
520
|
-
* Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,
|
|
521
|
-
* and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both
|
|
522
|
-
* metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row
|
|
523
|
-
* and results join back by harness/model via {@link harnessAxisOf} with no
|
|
524
|
-
* hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or
|
|
525
|
-
* its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every
|
|
526
|
-
* requested harness runs; `keepIncompatible` forces every pair verbatim.
|
|
527
|
-
*
|
|
528
|
-
* Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
|
|
529
|
-
* everything we care about" switch, identical in shape whether one harness or all.
|
|
530
|
-
*/
|
|
531
|
-
function expandProfileAxes(spec) {
|
|
532
|
-
const harnesses = spec.harnesses ?? CODING_HARNESSES;
|
|
533
|
-
if (harnesses.length === 0) throw new ValidationError("expandProfileAxes: no harnesses to sweep");
|
|
534
|
-
const baseModel = spec.base.model?.default;
|
|
535
|
-
const models = spec.models ?? (baseModel ? [baseModel] : []);
|
|
536
|
-
if (models.length === 0) throw new ValidationError("expandProfileAxes: no models to sweep — base profile has no model.default and none were supplied");
|
|
537
|
-
const out = [];
|
|
538
|
-
const seen = /* @__PURE__ */ new Set();
|
|
539
|
-
for (const harness of harnesses) {
|
|
540
|
-
const supported = spec.keepIncompatible ? models : models.filter((model) => harnessSupportsModel(harness, model));
|
|
541
|
-
const effective = supported.length > 0 ? supported : [HARNESS_NATIVE_MODEL];
|
|
542
|
-
for (const model of effective) {
|
|
543
|
-
const profile = {
|
|
544
|
-
...spec.base,
|
|
545
|
-
harness,
|
|
546
|
-
name: `${spec.base.name ?? "agent"}/${harness}/${model}`,
|
|
547
|
-
model: {
|
|
548
|
-
...spec.base.model,
|
|
549
|
-
default: model
|
|
550
|
-
},
|
|
551
|
-
metadata: {
|
|
552
|
-
...spec.base.metadata ?? {},
|
|
553
|
-
harness,
|
|
554
|
-
harnessModel: model
|
|
555
|
-
}
|
|
556
|
-
};
|
|
557
|
-
const id = agentProfileId(profile);
|
|
558
|
-
if (seen.has(id)) continue;
|
|
559
|
-
seen.add(id);
|
|
560
|
-
out.push(profile);
|
|
561
|
-
}
|
|
562
|
-
}
|
|
563
|
-
if (out.length === 0) throw new ValidationError(`expandProfileAxes: produced no profiles (harnesses=[${harnesses.join(", ")}], models=[${models.join(", ")}]).`);
|
|
564
|
-
return out;
|
|
565
|
-
}
|
|
566
|
-
/**
|
|
567
|
-
* Read the (harness, model) a matrix cell ran under, off a profile or a result row's
|
|
568
|
-
* profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
|
|
569
|
-
* wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
|
|
570
|
-
* this instead of recomputing an id (recomputing the wrong key is what broke the pivot
|
|
571
|
-
* in the hand-rolled copies).
|
|
572
|
-
*/
|
|
573
|
-
function harnessAxisOf(profile) {
|
|
574
|
-
const m = profile.metadata;
|
|
575
|
-
const harness = m?.harness;
|
|
576
|
-
const model = m?.harnessModel;
|
|
577
|
-
if (typeof harness === "string" && typeof model === "string") return {
|
|
578
|
-
harness,
|
|
579
|
-
model
|
|
580
|
-
};
|
|
581
|
-
}
|
|
582
|
-
/**
|
|
583
|
-
* Collision-resistant, path-safe, human-readable profile id for eval artifacts.
|
|
584
|
-
* Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
|
|
585
|
-
* keys, and directory names where two profiles must not collapse onto one row.
|
|
586
|
-
* The suffix is the first 64 bits of the behaviour hash, enough for ordinary
|
|
587
|
-
* eval matrices while keeping filenames readable.
|
|
588
|
-
*/
|
|
589
|
-
function agentProfileId(profile) {
|
|
590
|
-
return `${pathSafeProfileLabel(agentProfileDisplayLabel(profile)) ?? "profile"}-${agentProfileHash(profile).slice(0, 16)}`;
|
|
591
|
-
}
|
|
592
|
-
/**
|
|
593
|
-
* Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete
|
|
594
|
-
* model id because run records reject bare/missing model aliases.
|
|
595
|
-
*/
|
|
596
|
-
function agentProfileModelId(profile) {
|
|
597
|
-
const model = profile.model?.default?.trim();
|
|
598
|
-
if (!model) throw new ValidationError(`AgentProfile "${agentProfileDisplayLabel(profile) ?? "unnamed profile"}" has no model.default — cannot record eval run`);
|
|
599
|
-
return model;
|
|
600
|
-
}
|
|
601
|
-
function agentProfileDisplayLabel(profile) {
|
|
602
|
-
return profile.name?.trim() || profile.version?.trim() || void 0;
|
|
603
|
-
}
|
|
604
|
-
function pathSafeProfileLabel(label) {
|
|
605
|
-
return label?.trim().replace(/[^A-Za-z0-9._-]+/g, "-").replace(/-+/g, "-").replace(/^-|-$/g, "") || void 0;
|
|
606
|
-
}
|
|
607
|
-
function compact(input) {
|
|
608
|
-
const out = {};
|
|
609
|
-
for (const [key, value] of Object.entries(input)) if (value !== void 0) out[key] = value;
|
|
610
|
-
return out;
|
|
611
|
-
}
|
|
612
|
-
/**
|
|
613
|
-
* Deterministic behaviour identity for the canonical
|
|
614
|
-
* `@tangle-network/agent-interface` AgentProfile.
|
|
615
|
-
*
|
|
616
|
-
* `name` and `description` are labels and do not affect the hash. Profile
|
|
617
|
-
* `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
|
|
618
|
-
* and extensions do affect the hash. Resource array order is hash-bearing
|
|
619
|
-
* because mount order can change agent behaviour. Undefined fields are treated
|
|
620
|
-
* as absent; explicit `null` fields remain hash-bearing.
|
|
621
|
-
*/
|
|
622
|
-
function agentProfileHash(profile) {
|
|
623
|
-
const model = agentProfileModelId(profile);
|
|
624
|
-
const behaviour = {
|
|
625
|
-
...profile,
|
|
626
|
-
name: void 0,
|
|
627
|
-
description: void 0,
|
|
628
|
-
tags: profile.tags ? [...profile.tags].sort() : void 0,
|
|
629
|
-
model: compact({
|
|
630
|
-
...profile.model,
|
|
631
|
-
default: model
|
|
632
|
-
})
|
|
633
|
-
};
|
|
634
|
-
return createHash("sha256").update(JSON.stringify(canonicalize(behaviour))).digest("hex");
|
|
635
|
-
}
|
|
636
|
-
//#endregion
|
|
23
|
+
import { execFileSync } from "node:child_process";
|
|
637
24
|
//#region src/campaign/analyst-surface.ts
|
|
638
25
|
function buildTraceAnalystSurfaceDispatch(options) {
|
|
639
26
|
return async (surface, scenario, context) => {
|
|
@@ -3790,6 +3177,6 @@ function resolveWorktreePath(surface, worktreeDir) {
|
|
|
3790
3177
|
return verifyCodeSurface(surface, worktreeDir).path;
|
|
3791
3178
|
}
|
|
3792
3179
|
//#endregion
|
|
3793
|
-
export { planEvalFixtureRun as A,
|
|
3180
|
+
export { planEvalFixtureRun as A, LabeledScenarioStoreError as C, discoverEvalFixtures as D, neutralizationGate as E, buildTraceAnalystSurfaceDispatch as M, traceAnalystQualityJudge as N, loadEvalFixture as O, FsLabeledScenarioStore as S, rolloutArgumentDiff as T, renderScoreboardMarkdown as _, autoevalsScorerJudge as a, userStoryScoreboard as b, FileSearchLedger as c, validateSearchLedgerEvent as d, scoreDiscrimination as f, makePlaybackDispatch as g, runProfileMatrix as h, verifyCodeSurface as i, analyzeCrossSurfaceInteractions as j, loadEvalFixtureScenarios as k, SEARCH_LEDGER_SCHEMA as l, ProfileMatrixError as m, gitWorktreeAdapter as n, phoenixEvaluatorJudge as o, selectDiscriminative as p, resolveWorktreePath as r, isTransientTransportFailure as s, WorktreeAdapterError as t, openSearchLedger as u, scoreUserStory as v, classifyUngroundedLiterals as w, neutralizeText as x, scoreboardSummary as y };
|
|
3794
3181
|
|
|
3795
|
-
//# sourceMappingURL=campaign-
|
|
3182
|
+
//# sourceMappingURL=campaign-BYjBAypg.js.map
|