@tangle-network/agent-eval 0.150.1 → 0.161.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +178 -1
- package/README.md +7 -3
- package/dist/{active-curriculum-C4mk67HP.js → active-curriculum-CD5TU2yW.js} +3 -13
- package/dist/active-curriculum-CD5TU2yW.js.map +1 -0
- package/dist/{agent-profile-cell-BkcRDikH.d.ts → agent-profile-cell-CTOZJUuE.d.ts} +4 -2
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +19 -36
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/analyst/index.js.map +1 -1
- package/dist/{backend-integrity-DOCa_QrR.d.ts → backend-integrity-DxuQCu_A.d.ts} +4 -3
- package/dist/backend-integrity-DxuQCu_A.d.ts.map +1 -0
- package/dist/{benchmark-BtAWA8nT.d.ts → benchmark-CGPp-kDC.d.ts} +3 -3
- package/dist/{benchmark-BtAWA8nT.d.ts.map → benchmark-CGPp-kDC.d.ts.map} +1 -1
- package/dist/{benchmark-command-CAFwbH0L.js → benchmark-command-BVtaq_ve.js} +26 -31
- package/dist/benchmark-command-BVtaq_ve.js.map +1 -0
- package/dist/benchmarks/index.d.ts +6 -19
- package/dist/benchmarks/index.d.ts.map +1 -1
- package/dist/benchmarks/index.js +4 -4
- package/dist/benchmarks/index.js.map +1 -1
- package/dist/builder-eval/index.d.ts +3 -3
- package/dist/builder-eval/index.js +3 -3
- package/dist/campaign/index.d.ts +9 -9
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-CN_7xJdV.js → campaign-BSmOwskD.js} +77 -795
- package/dist/campaign-BSmOwskD.js.map +1 -0
- package/dist/{canonical-D-XsTQ6_.js → canonical-IL-Bu-14.js} +26 -2
- package/dist/canonical-IL-Bu-14.js.map +1 -0
- package/dist/{capture-fetch-BBVFzhkk.d.ts → capture-fetch-CqwsJkkG.d.ts} +3 -3
- package/dist/{capture-fetch-BBVFzhkk.d.ts.map → capture-fetch-CqwsJkkG.d.ts.map} +1 -1
- package/dist/{chat-client-2bVfrzhN.js → chat-client-DlMlAeYI.js} +5 -55
- package/dist/{chat-client-2bVfrzhN.js.map → chat-client-DlMlAeYI.js.map} +1 -1
- package/dist/chat-json-call-6g5sJobJ.js +53 -0
- package/dist/chat-json-call-6g5sJobJ.js.map +1 -0
- package/dist/cli.js +54 -19
- package/dist/cli.js.map +1 -1
- package/dist/{client-LIuo-KPv.js → client-CX7KqIdB.js} +3 -3
- package/dist/client-CX7KqIdB.js.map +1 -0
- package/dist/{client-kPQYT_56.d.ts → client-L9VVPkim.d.ts} +4 -4
- package/dist/{client-kPQYT_56.d.ts.map → client-L9VVPkim.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -27
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +14 -17
- package/dist/contract/index.js.map +1 -1
- package/dist/{counterfactual--bpysZF0.d.ts → counterfactual-BaFUWK3H.d.ts} +4 -4
- package/dist/{counterfactual--bpysZF0.d.ts.map → counterfactual-BaFUWK3H.d.ts.map} +1 -1
- package/dist/{counterfactual-lDfCx0Uz.js → counterfactual-D_VWavVm.js} +2 -2
- package/dist/{counterfactual-lDfCx0Uz.js.map → counterfactual-D_VWavVm.js.map} +1 -1
- package/dist/{dataset-CJjKqQfA.d.ts → dataset-DQqhOCPt.d.ts} +5 -4
- package/dist/{dataset-CJjKqQfA.d.ts.map → dataset-DQqhOCPt.d.ts.map} +1 -1
- package/dist/{default-registry-Cw0Ohdoj.d.ts → default-registry-G9CKMNkc.d.ts} +7 -8
- package/dist/{default-registry-Cw0Ohdoj.d.ts.map → default-registry-G9CKMNkc.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CEQWL9Hy.d.ts → define-agent-eval-Dx1JnPEa.d.ts} +26 -6
- package/dist/define-agent-eval-Dx1JnPEa.d.ts.map +1 -0
- package/dist/{define-agent-eval-rqNyVhVV.js → define-agent-eval-h-s-sI-v.js} +17 -11
- package/dist/define-agent-eval-h-s-sI-v.js.map +1 -0
- package/dist/{descriptive-B5MwKfbf.js → descriptive-jDOuI6mz.js} +22 -2
- package/dist/descriptive-jDOuI6mz.js.map +1 -0
- package/dist/{dspy-rlm-engine-DHI0WrUU.js → dspy-rlm-engine-DptEII26.js} +95 -15
- package/dist/dspy-rlm-engine-DptEII26.js.map +1 -0
- package/dist/{emitter-CPBAhxum.js → emitter-BpYFQPj4.js} +2 -18
- package/dist/emitter-BpYFQPj4.js.map +1 -0
- package/dist/{emitter-DGQGoLyj.d.ts → emitter-D_jYSGRd.d.ts} +4 -20
- package/dist/{emitter-DGQGoLyj.d.ts.map → emitter-D_jYSGRd.d.ts.map} +1 -1
- package/dist/{engine-BLzhNzoY.d.ts → engine-Cu5qD5Fc.d.ts} +9 -11
- package/dist/{engine-BLzhNzoY.d.ts.map → engine-Cu5qD5Fc.d.ts.map} +1 -1
- package/dist/{eval-campaign-C4jmuM-b.js → eval-campaign-BsXWL2-2.js} +17 -28
- package/dist/eval-campaign-BsXWL2-2.js.map +1 -0
- package/dist/{exact-types-ccQAyut1.d.ts → exact-types-qnexxJ1Z.d.ts} +2 -2
- package/dist/{exact-types-ccQAyut1.d.ts.map → exact-types-qnexxJ1Z.d.ts.map} +1 -1
- package/dist/{exec-y-DCLqK7.js → exec-D9WpA2p-.js} +2 -2
- package/dist/exec-D9WpA2p-.js.map +1 -0
- package/dist/experiment/index.d.ts +45 -11
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +30 -11
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-0MhuPArU.d.ts → experiment-tracker-DCO6Cz4s.d.ts} +2 -2
- package/dist/{experiment-tracker-0MhuPArU.d.ts.map → experiment-tracker-DCO6Cz4s.d.ts.map} +1 -1
- package/dist/{exporters-q9iL-2Jf.js → exporters-Df7TgHFv.js} +3 -3
- package/dist/exporters-Df7TgHFv.js.map +1 -0
- package/dist/{external-optimizer-contracts-DbLsm4Po.d.ts → external-optimizer-contracts-szBJ_1vh.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-DbLsm4Po.d.ts.map → external-optimizer-contracts-szBJ_1vh.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Bhmzngf-.js → external-optimizer-process-WosTBChy.js} +4 -4
- package/dist/{external-optimizer-process-Bhmzngf-.js.map → external-optimizer-process-WosTBChy.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-Dn90UqN2.js → external-optimizer-subprocess-BIWbHpgD.js} +10 -6
- package/dist/external-optimizer-subprocess-BIWbHpgD.js.map +1 -0
- package/dist/{failure-cluster-BLURuWG4.d.ts → failure-cluster-CXL8NbEw.d.ts} +3 -3
- package/dist/{failure-cluster-BLURuWG4.d.ts.map → failure-cluster-CXL8NbEw.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-DpTTjo0q.d.ts → feedback-trajectory-B3ZHaHV_.d.ts} +7 -7
- package/dist/{feedback-trajectory-DpTTjo0q.d.ts.map → feedback-trajectory-B3ZHaHV_.d.ts.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/{hf-dataset-XggBupCr.js → hf-dataset-D8_RNIis.js} +4 -4
- package/dist/{hf-dataset-XggBupCr.js.map → hf-dataset-D8_RNIis.js.map} +1 -1
- package/dist/hosted/index.d.ts +3 -14
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +2 -2
- package/dist/{index-CTKpu9ry.d.ts → index-CGtH1piv.d.ts} +48 -26
- package/dist/index-CGtH1piv.d.ts.map +1 -0
- package/dist/{index-BNPtkBPf.d.ts → index-D-IiQIBB.d.ts} +5 -10
- package/dist/index-D-IiQIBB.d.ts.map +1 -0
- package/dist/{index-IQccV3Ou.d.ts → index-D-V8gCs_.d.ts} +13 -90
- package/dist/index-D-V8gCs_.d.ts.map +1 -0
- package/dist/{index-B8Ui1mr1.d.ts → index-lfaSeKSD.d.ts} +18 -2
- package/dist/index-lfaSeKSD.d.ts.map +1 -0
- package/dist/index-vrJugRal.d.ts +1 -0
- package/dist/index.d.ts +68 -125
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +61 -112
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-BeT8KCgI.d.ts → insight-report-DRe8LB6d.d.ts} +4 -4
- package/dist/{insight-report-BeT8KCgI.d.ts.map → insight-report-DRe8LB6d.d.ts.map} +1 -1
- package/dist/{integrity-DysDBWDu.js → integrity-CyWSSoQS.js} +17 -4
- package/dist/integrity-CyWSSoQS.js.map +1 -0
- package/dist/{integrity-B0dZ96EO.d.ts → integrity-DUNX9Fao.d.ts} +3 -3
- package/dist/{integrity-B0dZ96EO.d.ts.map → integrity-DUNX9Fao.d.ts.map} +1 -1
- package/dist/{judge-calibration-DZkWrm5H.js → judge-calibration-zZjLz8hr.js} +2 -2
- package/dist/{judge-calibration-DZkWrm5H.js.map → judge-calibration-zZjLz8hr.js.map} +1 -1
- package/dist/{kind-factory-DmAa0h3K.js → kind-factory-DY8FdoXf.js} +3 -71
- package/dist/kind-factory-DY8FdoXf.js.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +3 -3
- package/dist/{ledger-core-DTae9rv_.js → ledger-core-BOzlRygb.js} +2 -2
- package/dist/{ledger-core-DTae9rv_.js.map → ledger-core-BOzlRygb.js.map} +1 -1
- package/dist/{llm-client-Bg32RW0j.js → llm-client-hgDieDNN.js} +53 -98
- package/dist/llm-client-hgDieDNN.js.map +1 -0
- package/dist/{llm-judge-CVq33oz1.js → llm-judge-BhasIPFT.js} +1136 -78
- package/dist/llm-judge-BhasIPFT.js.map +1 -0
- package/dist/{matrix-DrVnRp4G.d.ts → matrix-eXKRMHnL.d.ts} +74 -72
- package/dist/matrix-eXKRMHnL.d.ts.map +1 -0
- package/dist/meta-eval/index.d.ts +8 -6
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +9 -7
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-BV6tLVWl.js → mint-DfODW1KW.js} +3 -3
- package/dist/{mint-BV6tLVWl.js.map → mint-DfODW1KW.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +2 -8
- package/dist/multishot/golden/index.d.ts.map +1 -1
- package/dist/multishot/golden/index.js +56 -86
- package/dist/multishot/golden/index.js.map +1 -1
- package/dist/multishot/index.d.ts +11 -46
- package/dist/multishot/index.d.ts.map +1 -1
- package/dist/multishot/index.js +30 -83
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-DJWAXLms.js → opencode-sqlite-eK6HW6dr.js} +2 -6
- package/dist/{opencode-sqlite-DJWAXLms.js.map → opencode-sqlite-eK6HW6dr.js.map} +1 -1
- package/dist/pipelines/index.d.ts +5 -5
- package/dist/pipelines/index.js +3 -3
- package/dist/{pre-registration-zFSLEiFU.d.ts → pre-registration-CzFCcwYk.d.ts} +55 -40
- package/dist/pre-registration-CzFCcwYk.d.ts.map +1 -0
- package/dist/pre-registration-KN9jkh58.js +110 -0
- package/dist/pre-registration-KN9jkh58.js.map +1 -0
- package/dist/{produced-state-jfk8Du3b.js → produced-state-DZ89riy5.js} +8 -8
- package/dist/produced-state-DZ89riy5.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +31 -5
- package/dist/profile-cell.js.map +1 -1
- package/dist/{promotion-policy-DLOUkYhI.d.ts → promotion-policy-DtnOIZvk.d.ts} +2 -2
- package/dist/{promotion-policy-DLOUkYhI.d.ts.map → promotion-policy-DtnOIZvk.d.ts.map} +1 -1
- package/dist/{query-Di7eEQ79.js → query-CHmMP42p.js} +20 -11
- package/dist/query-CHmMP42p.js.map +1 -0
- package/dist/{query-CJ_DX8vl.d.ts → query-DxPYqpmT.d.ts} +10 -4
- package/dist/query-DxPYqpmT.d.ts.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/{registry-BQwrSYpC.d.ts → registry-8You7OK1.d.ts} +5 -7
- package/dist/{registry-BQwrSYpC.d.ts.map → registry-8You7OK1.d.ts.map} +1 -1
- package/dist/{release-confidence-BknrpBnO.js → release-confidence-DKfD2RYU.js} +28 -14
- package/dist/release-confidence-DKfD2RYU.js.map +1 -0
- package/dist/{release-confidence-4XrqlpFD.d.ts → release-confidence-Dqt0NFep.d.ts} +7 -6
- package/dist/release-confidence-Dqt0NFep.d.ts.map +1 -0
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +3 -3
- package/dist/{researcher-DJnoUE8c.d.ts → researcher-Cz565b7D.d.ts} +34 -21
- package/dist/researcher-Cz565b7D.d.ts.map +1 -0
- package/dist/{reward-hacking-62tojkQd.d.ts → reward-hacking-MBf7qpSB.d.ts} +2 -2
- package/dist/{reward-hacking-62tojkQd.d.ts.map → reward-hacking-MBf7qpSB.d.ts.map} +1 -1
- package/dist/{reward-hacking-DKI9T52l.js → reward-hacking-t4lB1yt8.js} +3 -3
- package/dist/{reward-hacking-DKI9T52l.js.map → reward-hacking-t4lB1yt8.js.map} +1 -1
- package/dist/rl.d.ts +11 -42
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +41 -24
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +3 -3
- package/dist/rollout/index.js +7 -7
- package/dist/{rollout-ytVQ7WT8.js → rollout-Dm2tSdiQ.js} +6 -6
- package/dist/{rollout-ytVQ7WT8.js.map → rollout-Dm2tSdiQ.js.map} +1 -1
- package/dist/{rubric-predictive-validity-Cwwyd7ah.js → rubric-predictive-validity-CK8SCOg-.js} +6 -16
- package/dist/rubric-predictive-validity-CK8SCOg-.js.map +1 -0
- package/dist/{rubric-predictive-validity-C7LnNvF2.d.ts → rubric-predictive-validity-CxycqzX5.d.ts} +4 -3
- package/dist/rubric-predictive-validity-CxycqzX5.d.ts.map +1 -0
- package/dist/{run-record-D2lDdSAz.js → run-record-BC0ebuRP.js} +2 -2
- package/dist/{run-record-D2lDdSAz.js.map → run-record-BC0ebuRP.js.map} +1 -1
- package/dist/{run-record-DVV82Gwh.d.ts → run-record-VVy4T9OW.d.ts} +3 -3
- package/dist/{run-record-DVV82Gwh.d.ts.map → run-record-VVy4T9OW.d.ts.map} +1 -1
- package/dist/{schema-BtVldJ3T.d.ts → schema-Bjgdsn73.d.ts} +2 -4
- package/dist/{schema-BtVldJ3T.d.ts.map → schema-Bjgdsn73.d.ts.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-BzWDXhOR.d.ts} +2 -5
- package/dist/schema-BzWDXhOR.d.ts.map +1 -0
- package/dist/{schema-C6DW4ZHR.js → schema-C1aaAxTf.js} +2 -2
- package/dist/schema-C1aaAxTf.js.map +1 -0
- package/dist/{schema-CRhEY1SO.js → schema-k6ZBftVv.js} +2 -8
- package/dist/{schema-CRhEY1SO.js.map → schema-k6ZBftVv.js.map} +1 -1
- package/dist/{semantic-concept-judge-laMCnTLn.js → semantic-concept-judge-BSkKKHeq.js} +14 -38
- package/dist/semantic-concept-judge-BSkKKHeq.js.map +1 -0
- package/dist/{sequential-eprocess-CbUt2htw.js → sequential-eprocess-D1jKoihe.js} +49 -2
- package/dist/sequential-eprocess-D1jKoihe.js.map +1 -0
- package/dist/{sequential-C458DXNf.js → sequential-rYW-Ophm.js} +41 -16
- package/dist/sequential-rYW-Ophm.js.map +1 -0
- package/dist/{series-convergence-BxKEgBwA.d.ts → series-convergence-D9WgpXGi.d.ts} +2 -2
- package/dist/{series-convergence-BxKEgBwA.d.ts.map → series-convergence-D9WgpXGi.d.ts.map} +1 -1
- package/dist/{server-dIWwF3j_.js → server-BtFd4uzB.js} +19 -42
- package/dist/server-BtFd4uzB.js.map +1 -0
- package/dist/{skillopt-optimization-method-DLeUcK-K.js → skillopt-optimization-method-DbaekMcn.js} +794 -8
- package/dist/skillopt-optimization-method-DbaekMcn.js.map +1 -0
- package/dist/{skillopt-optimization-method-BDD_o1xE.d.ts → skillopt-optimization-method-x7TTF23P.d.ts} +20 -7
- package/dist/{skillopt-optimization-method-BDD_o1xE.d.ts.map → skillopt-optimization-method-x7TTF23P.d.ts.map} +1 -1
- package/dist/{statistical-heldout-_woZ9q9j.d.ts → statistical-heldout-Cy3EhjlC.d.ts} +21 -8
- package/dist/statistical-heldout-Cy3EhjlC.d.ts.map +1 -0
- package/dist/{steps-AmkT-GIM.d.ts → steps-CiNVJry_.d.ts} +2 -17
- package/dist/steps-CiNVJry_.d.ts.map +1 -0
- package/dist/{store-CT9YIIve.d.ts → store-B06JdC56.d.ts} +2 -2
- package/dist/{store-CT9YIIve.d.ts.map → store-B06JdC56.d.ts.map} +1 -1
- package/dist/{store-otlp-CDYWW_8N.js → store-otlp-C_Rq5I4D.js} +2 -2
- package/dist/{store-otlp-CDYWW_8N.js.map → store-otlp-C_Rq5I4D.js.map} +1 -1
- package/dist/{store-tool-spans-Br2_IUhm.d.ts → store-tool-spans-DPUG7UUY.d.ts} +6 -6
- package/dist/{store-tool-spans-Br2_IUhm.d.ts.map → store-tool-spans-DPUG7UUY.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CykkbOlv.js → store-tool-spans-Dlh9vkFK.js} +3 -3
- package/dist/{store-tool-spans-CykkbOlv.js.map → store-tool-spans-Dlh9vkFK.js.map} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-Blysd6Z2.js → summary-report-BI5hUtvK.js} +7 -17
- package/dist/summary-report-BI5hUtvK.js.map +1 -0
- package/dist/{summary-report-B__Y5ub3.d.ts → summary-report-CC07PhEL.d.ts} +6 -5
- package/dist/summary-report-CC07PhEL.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +27 -18
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +104 -26
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{task-failure-attributes--ZTP3tYO.js → task-failure-attributes-DTl-7-Kw.js} +3 -3
- package/dist/{task-failure-attributes--ZTP3tYO.js.map → task-failure-attributes-DTl-7-Kw.js.map} +1 -1
- package/dist/{tool-groups-B4tqh8jB.d.ts → tool-groups-Ci8i9ErB.d.ts} +3 -3
- package/dist/tool-groups-Ci8i9ErB.d.ts.map +1 -0
- package/dist/{tool-waste-BDdBZG1F.js → tool-waste-BqzmVdJk.js} +4 -4
- package/dist/{tool-waste-BDdBZG1F.js.map → tool-waste-BqzmVdJk.js.map} +1 -1
- package/dist/{tool-waste-DjRDEsuI.d.ts → tool-waste-Dro0gJi3.d.ts} +4 -4
- package/dist/{tool-waste-DjRDEsuI.d.ts.map → tool-waste-Dro0gJi3.d.ts.map} +1 -1
- package/dist/trace-repair/index.d.ts +4 -77
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +5 -15
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +13 -23
- package/dist/traces.d.ts.map +1 -1
- package/dist/traces.js +9 -20
- package/dist/traces.js.map +1 -1
- package/dist/{trajectory-YC15QDYQ.d.ts → trajectory-Bi157Gun.d.ts} +3 -3
- package/dist/{trajectory-YC15QDYQ.d.ts.map → trajectory-Bi157Gun.d.ts.map} +1 -1
- package/dist/trajectory-replay/index.d.ts +5 -5
- package/dist/trajectory-replay/index.js +5 -5
- package/dist/{provenance-oA4-zUqm.d.ts → transient-failure-DKF5Mofa.d.ts} +468 -13
- package/dist/transient-failure-DKF5Mofa.d.ts.map +1 -0
- package/dist/types-B3jzCp0p.js.map +1 -1
- package/dist/{types-CLAwnY-L.d.ts → types-BPb2Kf_C2.d.ts} +3 -3
- package/dist/types-BPb2Kf_C2.d.ts.map +1 -0
- package/dist/types-Bfk0uxRj.d.ts +443 -0
- package/dist/types-Bfk0uxRj.d.ts.map +1 -0
- package/dist/{types-DdFNuyxQ.d.ts → types-D4s7Z6nq.d.ts} +30 -6
- package/dist/types-D4s7Z6nq.d.ts.map +1 -0
- package/dist/{types-B2NsbrNy.d.ts → types-D9ssmxKL.d.ts} +3 -3
- package/dist/{types-B2NsbrNy.d.ts.map → types-D9ssmxKL.d.ts.map} +1 -1
- package/dist/{types-yLK8gXE9.d.ts → types-DeIUdzNd.d.ts} +160 -10
- package/dist/types-DeIUdzNd.d.ts.map +1 -0
- package/dist/{verdict-BndeTAh_.js → verdict-B0xltqu6.js} +2 -2
- package/dist/{verdict-BndeTAh_.js.map → verdict-B0xltqu6.js.map} +1 -1
- package/dist/verdict-cache-CdVVTVmn.js +88 -0
- package/dist/verdict-cache-CdVVTVmn.js.map +1 -0
- package/dist/wire/index.d.ts +21 -111
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +2 -2
- package/docs/adapters-observability.md +9 -23
- package/docs/building-doctrine.md +3 -3
- package/docs/campaign-proposers.md +54 -15
- package/docs/concepts.md +3 -4
- package/docs/design/statistics-decisions.md +89 -1
- package/docs/eval-surface-map.md +14 -0
- package/docs/experiment.md +19 -2
- package/docs/feedback-trajectories.md +1 -1
- package/docs/multishot-golden-records.md +4 -4
- package/docs/public-api.md +1616 -0
- package/docs/research-report-methodology.md +1 -1
- package/docs/search-history-receipts.md +39 -1
- package/docs/trace-analysis.md +1 -1
- package/docs/trace-repair-admission.md +1 -1
- package/docs/trace-repair-continuation.md +1 -1
- package/docs/verdicts.md +24 -0
- package/docs/wire-protocol.md +1 -1
- package/package.json +6 -2
- package/dist/active-curriculum-C4mk67HP.js.map +0 -1
- package/dist/agent-profile-cell-BkcRDikH.d.ts.map +0 -1
- package/dist/backend-integrity-DOCa_QrR.d.ts.map +0 -1
- package/dist/benchmark-command-CAFwbH0L.js.map +0 -1
- package/dist/campaign-CN_7xJdV.js.map +0 -1
- package/dist/canonical-D-XsTQ6_.js.map +0 -1
- package/dist/client-LIuo-KPv.js.map +0 -1
- package/dist/define-agent-eval-CEQWL9Hy.d.ts.map +0 -1
- package/dist/define-agent-eval-rqNyVhVV.js.map +0 -1
- package/dist/descriptive-B5MwKfbf.js.map +0 -1
- package/dist/dspy-rlm-engine-DHI0WrUU.js.map +0 -1
- package/dist/emitter-CPBAhxum.js.map +0 -1
- package/dist/eval-campaign-C4jmuM-b.js.map +0 -1
- package/dist/exec-y-DCLqK7.js.map +0 -1
- package/dist/exporters-q9iL-2Jf.js.map +0 -1
- package/dist/external-optimizer-subprocess-Dn90UqN2.js.map +0 -1
- package/dist/index-B8Ui1mr1.d.ts.map +0 -1
- package/dist/index-BNPtkBPf.d.ts.map +0 -1
- package/dist/index-C1ravkGA.d.ts +0 -1
- package/dist/index-CTKpu9ry.d.ts.map +0 -1
- package/dist/index-IQccV3Ou.d.ts.map +0 -1
- package/dist/integrity-DysDBWDu.js.map +0 -1
- package/dist/kind-factory-DmAa0h3K.js.map +0 -1
- package/dist/llm-client-Bg32RW0j.js.map +0 -1
- package/dist/llm-judge-CVq33oz1.js.map +0 -1
- package/dist/matrix-DrVnRp4G.d.ts.map +0 -1
- package/dist/pre-registration-DakwTRXk.js +0 -96
- package/dist/pre-registration-DakwTRXk.js.map +0 -1
- package/dist/pre-registration-zFSLEiFU.d.ts.map +0 -1
- package/dist/produced-state-jfk8Du3b.js.map +0 -1
- package/dist/provenance-oA4-zUqm.d.ts.map +0 -1
- package/dist/query-CJ_DX8vl.d.ts.map +0 -1
- package/dist/query-Di7eEQ79.js.map +0 -1
- package/dist/release-confidence-4XrqlpFD.d.ts.map +0 -1
- package/dist/release-confidence-BknrpBnO.js.map +0 -1
- package/dist/researcher-DJnoUE8c.d.ts.map +0 -1
- package/dist/rubric-predictive-validity-C7LnNvF2.d.ts.map +0 -1
- package/dist/rubric-predictive-validity-Cwwyd7ah.js.map +0 -1
- package/dist/schema-C6DW4ZHR.js.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/semantic-concept-judge-laMCnTLn.js.map +0 -1
- package/dist/sequential-C458DXNf.js.map +0 -1
- package/dist/sequential-eprocess-CbUt2htw.js.map +0 -1
- package/dist/server-dIWwF3j_.js.map +0 -1
- package/dist/skillopt-optimization-method-DLeUcK-K.js.map +0 -1
- package/dist/statistical-heldout-_woZ9q9j.d.ts.map +0 -1
- package/dist/steps-AmkT-GIM.d.ts.map +0 -1
- package/dist/summary-report-B__Y5ub3.d.ts.map +0 -1
- package/dist/summary-report-Blysd6Z2.js.map +0 -1
- package/dist/tool-groups-B4tqh8jB.d.ts.map +0 -1
- package/dist/types-CLAwnY-L.d.ts.map +0 -1
- package/dist/types-DdFNuyxQ.d.ts.map +0 -1
- package/dist/types-jUBXJ7Iz.d.ts +0 -884
- package/dist/types-jUBXJ7Iz.d.ts.map +0 -1
- package/dist/types-yLK8gXE9.d.ts.map +0 -1
- package/dist/verdict-cache-mZf5FEiY.js +0 -107
- package/dist/verdict-cache-mZf5FEiY.js.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,7 +4,176 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
-
## [
|
|
7
|
+
## [0.161.0] — 2026-08-21
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- One raw-finding wire codec for both languages (#636, supersedes #567, #579, #606). `decodeRawFindingArray(value)` (TypeScript, `/analyst`) and `decode_raw_finding_array(value)` (Python, `agent_eval_rpc.finding_codec`) decode a findings array under one set of rules and return the accepted rows plus a diagnostic per refused row — index, field path, rejection code, message. Python validates against `finding_contract.json`, which `pnpm run contract:finding` generates from the TypeScript schema (`z.toJSONSchema` plus the subject grammar as acceptance patterns) and CI checks byte-for-byte, so neither side keeps a hand-maintained copy of the other's schema. A shared corpus in `tests/fixtures/finding-codec/` runs in both CI lanes and asserts the same accepted rows and the same rejection paths and codes.
|
|
12
|
+
- `FINDING_SUBJECT_PATTERNS` and `isFindingSubject()`: the subject grammar as portable acceptance patterns, proven equal to `parseFindingSubject` over every kind's valid and invalid forms. Python now enforces the subject grammar it previously ignored, so it can no longer report success for a row TypeScript rejects.
|
|
13
|
+
|
|
14
|
+
### Fixed
|
|
15
|
+
|
|
16
|
+
- A findings submission can no longer become a silent empty result. Measured on the previous release: a list handed to the bridge raised an unhelpful "must be a non-empty trimmed string" and discarded the completed investigation, while the Python repr of that same list parsed to zero rows and reported success. A list or mapping is now canonical-JSON encoded before it crosses the string boundary; a repr, unparseable text, or any non-array type raises with the type it received. Only an explicitly empty array is an empty result. `_SAFE_FIELD_DEFAULTS["findings_json"] = "[]"` is deleted, and an unrecoverable findings field is now reported by name instead of defaulting to an empty array.
|
|
17
|
+
- The generic repair prompt no longer carries CodeTrace's "return [] if the answer reports no incorrect steps" rule, which erased valid factual findings from a generic analysis; it asks for citable claims and returns `[]` only when the answer contains none. CodeTrace keeps its own prompt. The one declared repair turn now receives the exact row defects, and a repair failure preserves both the defects and the prose answer.
|
|
18
|
+
- Rejected rows travel in `runtime.rejectedFindings` from Python under the same key TypeScript already used, so a caller sees per-row diagnostics from both sides rather than a bare count from one.
|
|
19
|
+
|
|
20
|
+
### Removed
|
|
21
|
+
|
|
22
|
+
- `coerceToFindingRows` (no consumer outside its own test). It promoted a single object to a one-row array and turned any unrecognized value into `[]` — both widened what TypeScript accepted past what Python does, which is how one language reported findings the other dropped. `decodeRawFindingArray` replaces it and reports what it refused.
|
|
23
|
+
|
|
24
|
+
---
|
|
25
|
+
|
|
26
|
+
---
|
|
27
|
+
|
|
28
|
+
## [0.160.0] — 2026-08-21
|
|
29
|
+
|
|
30
|
+
### Removed
|
|
31
|
+
|
|
32
|
+
- Every Eval-owned paid model transport (#539). Agent Eval owns comparison, scoring, and durable evidence; it no longer executes a paid model, accepts a provider URL, or holds a credential. The caller supplies a `ChatClient`, and on Agent Runtime `profileChatClient` / `profileOptimizerModelCall` are that transport, so exact `AgentProfile` identity, retries, usage, cache accounting, and interruption safety stop being optional.
|
|
33
|
+
- `createChatClient` loses its `router`, `direct-provider`, and `cli-bridge` variants. `custom`, `sandbox-sdk`, and `mock` remain, and `ChatTransport` narrows to those three.
|
|
34
|
+
- The root barrel no longer exports `callLlm`, `callLlmJson`, `LlmClient`, `LlmClientOptions`, `assertLlmRoute`, `LlmRouteRequirements`, or `probeLlm`. `assertLlmRoute` and `probeLlm` are deleted outright: the caller holds the endpoint, so the caller owns both the route check and the reachability probe. The canonical contract stays public — `LlmCallRequest`, `LlmCallResult` (including `logprobs`, `toolCalls`, `servedModel`), `LlmMessage`, `LlmUsage`, `costReceiptFromLlm`, `costReceiptFromLlmError`, `maximumChargeForLlmRequest`, `isTransientLlmError`, `stripFencedJson`.
|
|
35
|
+
- `createOpenAiCompatibleExecutionOwner` is gone from `/campaign`. Agent Runtime already owns that role with `profileOptimizerModelCall`, which executes one exact `AgentProfile` and reports profile-digest evidence; two owners for one role was the defect. `examples/_shared/openai-compatible-owner.ts` is the caller-side reference implementation, and it is example code, not a published export.
|
|
36
|
+
- `multishot/router.ts` is deleted with `routerCompletion`, `requireRouterApiKey`, and `defaultRouterBaseUrl`. `runMultishot`, `runMultishotMatrix`, and `runJudge` now require a caller-supplied `MultishotTransport`; `JudgeConfig.transport` is required and `JUDGE_MODEL` is no longer read from the environment. `MultishotToolExecutor` receives `{ transport, signal }` instead of `{ apiKey, baseUrl, signal }`, and the optional `toolTransport` names the leg the built-in delegate tools run on. `estimateRouterCost` is now `estimateMultishotCost` in `multishot/cost.ts`.
|
|
37
|
+
- `preflightModels` and `assertModelsServed` take `request: ModelEndpointRequest` instead of `baseUrl` and `apiKey`. Agent Eval asks for a `list-models` or a `probe` check and reads the `Response`, so status, the provider's own `error.message`, `budgetExhausted`, and served-model substitution stay exactly as detectable as before.
|
|
38
|
+
- `runIntentMatchJudge`, `runSemanticConceptJudge`, `handleJudge`, `dispatchRpc`, and `createApp` take `chat: ChatClient` (plus optional `pricing`) instead of `llm: LlmClientOptions`. `/v1/judge` refuses with `llm_not_configured` (503) when no transport is configured, which replaces the old route assertion.
|
|
39
|
+
- The internal OpenAI-compatible client has no default endpoint. `DEFAULT_BASE_URL = 'https://router.tangle.tools/v1'` is deleted and `baseUrl` is required, so a misconfigured caller fails loudly instead of silently billing the public router — the failure `assertLlmRoute` existed to catch, now unrepresentable.
|
|
40
|
+
- `runEvalCampaign` takes `chatFactory: (wiring: CampaignChatWiring) => ChatClient` instead of `llmOpts`, and `CampaignRunContext.chat` replaces `ctx.llmOpts`. The campaign passes each run's `rawSink` and `runId` into the factory, so a transport that binds them still satisfies `assertRunCaptured`'s raw-coverage check. The campaign fingerprint now folds a caller-declared `executionRef` where it previously folded the base URL and provider it can no longer see.
|
|
41
|
+
|
|
42
|
+
### Added
|
|
43
|
+
|
|
44
|
+
- `paidJsonChat` collapses the three hand-rolled copies of "reserve the priced maximum, call the transport with a stable call id, settle the receipt, parse the JSON answer" that the two judges and the wire judge endpoint each carried. A malformed answer keeps its settled receipt: the call completed and was billed, so the spend stays known rather than becoming unknown.
|
|
45
|
+
- `LlmChargeBounds`: the narrow bound inputs `maximumChargeForLlmRequest` actually reads, so a caller can price a request without naming a transport options type.
|
|
46
|
+
- `examples/_shared/openai-compatible-owner.ts` exposes one OpenAI-compatible endpoint two ways — `openAiCompatibleChatClient` for judges and workers, `openAiCompatibleExecutionOwner` for the optimizer surface — as the reference for what caller-owned execution looks like.
|
|
47
|
+
|
|
48
|
+
### Changed
|
|
49
|
+
|
|
50
|
+
- The `agent-eval` binary is the one place in the package that reads a provider credential, and it is documented as such. `agent-eval serve` / `rpc` / `rpc-batch` build their own `ChatClient` from `AGENT_EVAL_LLM_*` (or the `OPENAI_*` / `TANGLE_*` equivalents) inside `src/cli-config.ts`. Both a base URL and a key are required; a half-configured server refuses instead of calling an unintended endpoint.
|
|
51
|
+
|
|
52
|
+
## [0.159.1] — 2026-08-21
|
|
53
|
+
|
|
54
|
+
### Added
|
|
55
|
+
|
|
56
|
+
- `pnpm check:canonical-json` (#646, part 3 of 3), wired into `verify:package`. The gate reads every function under `src/` and fails the ones that sort object keys AND serialize in the same body — the shape of a hand-rolled canonical-JSON encoder. Eleven such copies existed before this arc and disagreed on `undefined`-valued keys, `Date`, and integer-like key order, so one value hashed differently depending on which copy ran. Sorting counts through a module-local helper, so splitting the sort into `JSON.stringify(sortKeys(v))` does not evade the gate; a key-SET check that sorts and compares is not reported, because the sorted list is never iterated to build output. An explicit allowlist names the three private legacy verifiers that must keep the retired bytes, and a waiver that matches nothing fails the gate so a stale entry cannot hide a new copy. `CLAUDE.md` names `src/ledger-core/canonical.ts` as the one home.
|
|
57
|
+
|
|
58
|
+
---
|
|
59
|
+
|
|
60
|
+
## [0.159.0] — 2026-08-21
|
|
61
|
+
|
|
62
|
+
### Changed
|
|
63
|
+
|
|
64
|
+
- Durable digests name their scheme (#646, part 2 of 3). A signed `HypothesisManifest`, a `SealedExperiment`, and an `AgentProfileCell` id are each written once and verified later, possibly by a different release, so each now records the encoder that produced it and verification selects the encoder from the record. `signManifest` writes `algo: 'sha256-rfc8785'` and `sealExperiment` writes the same tag; an `AgentProfileCell` id is now `agent-profile-cell:sha256-rfc8785:<digest>`. Records written under the previous key-sorted `JSON.stringify` scheme — tagged `'sha256-content'`, carrying no `algo`, or carrying the bare `agent-profile-cell:sha256:` prefix — still verify. Each legacy encoder is private to the module that verifies with it and is unreachable from any path that writes a digest; `docs/experiment.md` records when they can be retired. An `algo` this release does not recognize is refused rather than read as valid.
|
|
65
|
+
- `hashJson`, `agentProfileHash`, and the analyst-benchmark receipt digests route through `ledger-core/canonical`. Benchmark receipts are digested before they are written and re-digested when they are read back, so they digest the JSON document form: the new exported `jsonDocument()` states that rule — a JSON file cannot carry `undefined`, so an undefined-valued key is absent — while every other ambiguous value stays refused, unlike `JSON.stringify`, which turns `NaN` into `null`. Analyst-benchmark receipts need no algorithm tag: the run identity that gates every resume already pins `implementationSha256`, a digest over the implementation sources including the encoder itself, so a receipt written by another release is refused before any digest is compared.
|
|
66
|
+
- `agentProfileHash` is byte-identical to the digest the previous encoder produced, and an AgentRx case definition no longer carries `undefined`-valued metadata keys — both pinned by tests.
|
|
67
|
+
|
|
68
|
+
### Removed
|
|
69
|
+
|
|
70
|
+
- `canonicalize` is no longer exported (root and `./experiment`). It was the key-sorting half of the retired digest scheme; `hashJson` canonicalizes internally, and a caller that needs the encoding directly uses `canonicalString` from `./ledger-core`. `manifestContentDigest` is exported in its place: the one synchronous manifest digest, selected by `algo`, that `sequentialPairedGate` now shares instead of re-implementing the rule.
|
|
71
|
+
|
|
72
|
+
---
|
|
73
|
+
|
|
74
|
+
## [0.158.0] — 2026-08-21
|
|
75
|
+
|
|
76
|
+
### Fixed
|
|
77
|
+
|
|
78
|
+
- Every result-bearing random draw is seeded (#411, criterion 2). Five byte-identical private copies of `mulberry32` lived beside the canonical one, in `meta-eval/rubric-predictive-validity.ts`, `rl/active-curriculum.ts`, `rl/adaptation-eval.ts`, `summary-report.ts`, and `promotion-gate.ts`; four fell back to `Math.random` when the caller passed no seed, so the rubric verdict, the Thompson allocation, the adaptation-curve intervals, and the reported posterior were silently non-reproducible. `meta-eval/correlation-study.ts` called `Math.random` directly inside its bootstrap and offered no seed at all. All six now route through `makeRng`, which derives the seed from the observations when the caller supplies none, so the same input reproduces the same interval; `correlationStudy` gains the `seed` option its sibling already had. `Math.random` references in `src/` fall from 19 to 10, and every remaining one produces an identifier or a retry delay that no reported number reads. The seat-by-seat audit table is in `docs/design/statistics-decisions.md`.
|
|
79
|
+
- Two documentation fences imported `@tangle-network/agent-eval/../src/trace-repair`, which cannot resolve; they now import `@tangle-network/agent-eval/trace-repair`. Three fences that are not TypeScript — an object fragment, a call elided as `{ ... }`, and a method sketch — are marked `text`.
|
|
80
|
+
|
|
81
|
+
### Changed
|
|
82
|
+
|
|
83
|
+
- Test classification for #411 criterion 4, recorded in `docs/design/statistics-decisions.md`: the no-throw assertions that discarded a returned value now assert the value; the ones that pin a void guard's accept case are kept, because for a fail-closed gate "does not refuse a legal input" is the assertion. The five module mocks each carry a one-line justification in their file header, and the six conditional skips are tabulated with the environment variable or platform that gates them.
|
|
84
|
+
|
|
85
|
+
---
|
|
86
|
+
|
|
87
|
+
## [0.157.0] — 2026-08-21
|
|
88
|
+
|
|
89
|
+
### Removed
|
|
90
|
+
|
|
91
|
+
- 86 published value exports that no consumer binds (#411, criterion 3). Every subpath's named value exports were enumerated mechanically and checked against 31 repositories that depend on this package (each on its default branch, commits recorded in `docs/public-api.md`), this package's own CLI, wire server, examples, tests, and Markdown front doors. A symbol with no evidence in ANY channel left the barrel; 11 whose declaration nothing referenced were deleted outright, and 75 that their own module still uses lost only the `export` keyword. A symbol whose only reference is a test, a doc, a script, or a second module was kept and stays listed as `none` in the census — the census names the five blind spots that make a `none` uncertain, and an uncertain `none` is kept, never deleted. Notable removals: `BUILTIN_RUBRICS` (`./wire`), `otelRunCompleteHook` (`./traces`), `FINDING_SUBJECT_GRAMMAR_PROMPT` and the analyst adapter factories (`./analyst`), the delegate-tool defaults (`./multishot`), and the `EvalRun*` schemas (`./hosted`).
|
|
92
|
+
|
|
93
|
+
### Added
|
|
94
|
+
|
|
95
|
+
- `docs/public-api.md` and `pnpm api:census`: every published value export with the consumer that justifies it, classified `production`, `planned`, or `none`, with the evidence for each row and the five blind spots that bound the classification (dynamic imports and namespace binds, repositories outside the sweep, pinned versions, the wire and RPC surfaces, string-keyed dispatch). Consumer evidence lives in `scripts/public-api-consumers.json`, stamped with the commit of every repository read. The census runs on demand: it is a dated reading, not a CI gate.
|
|
96
|
+
---
|
|
97
|
+
|
|
98
|
+
## [0.156.0] — 2026-08-21
|
|
99
|
+
|
|
100
|
+
### Added
|
|
101
|
+
|
|
102
|
+
- Logprob-expectation judge scoring (#637). `llmJudge({ scoring: { method: 'expectation', whenUnavailable } })` asks the provider for the log probabilities of the score token and returns the expected grade over the integer grades the model considered, so two answers that both sample `8` separate by how much mass sat on `7` versus `9`. It requires `scale: 'ten'` — an integer grade is one token and a `[0,1]` float is not — and refuses a grade that did not land in exactly one token instead of approximating it. `whenUnavailable` decides the provider-returns-nothing case: `'fail'` throws, `'sampled'` reads the emitted grade. `JudgeScore` gains `scoringMethod` (what actually produced the number, so a run that fell back reads `'sampled'`) and `distribution` (probability mass per grade). Panels are unchanged; `ensembleJudge` consumes the composite either way. No new dependency: the technique is a scoring-loop change, not a package.
|
|
103
|
+
- `LlmCallRequest.logprobs: { topLogprobs }` sends `logprobs: true` with `top_logprobs`, and `LlmCallResult.logprobs` carries the parsed per-token window from `choices[0].logprobs.content`, or `null` when the provider returned none. A provider that ignores the field is a fact the caller can read, never an inferred distribution. `wrapLlmClient` forwards both, so every `ChatClient` transport carries them.
|
|
104
|
+
---
|
|
105
|
+
|
|
106
|
+
## [0.155.0] — 2026-08-21
|
|
107
|
+
|
|
108
|
+
### Added
|
|
109
|
+
|
|
110
|
+
- `runOptimization({ searchLedger })` and `selfImprove({ searchLedger })` record the candidate search into the canonical `SearchLedger` and return a bounded `searchHistory` receipt (#633). The loop emits the plan (slots = generations x populationSize, one candidate-generation operation per generation, one selection operation, one task per designed scenario-replicate cell), one registration per candidate carrying the exact parent surface it mutated, one attempt per scored cell with the cell's own outcome and accounting, one decision per candidate, and the terminal event. `FileSearchLedger` now has a first-party caller in its own package. The terminal event is appended only when canonical replay accounts for the whole planned denominator, so an interrupted or partly unscored search reports the gap instead of claiming a closed search.
|
|
111
|
+
- `search-plan-extended`: a rolling search appends candidate slots and operations to an existing plan instead of opening a second ledger. Replay merges the first plan with every extension, the generation invariant continues across rounds, and the planless refusal is unchanged. The planned task denominator stays frozen.
|
|
112
|
+
- `gepaOptimizationMethod({ searchLedger: { identity } })` records GEPA's own candidate population — its parent graph and per-scenario selection scores — into the same ledger through `recordCandidatePopulationSearch()`. `compareOptimizationMethods({ searchHistoryPolicy: 'require-complete' })` now accepts a first-party method, which is what `docs/search-history-receipts.md` promised.
|
|
113
|
+
|
|
114
|
+
---
|
|
115
|
+
|
|
116
|
+
## [0.154.0] — 2026-08-21
|
|
117
|
+
|
|
118
|
+
### Added
|
|
119
|
+
|
|
120
|
+
- Sequential state reconstruction after a process restart (#411, criterion 2). `EProcessState` now carries the running sums `sumX` and `varSum` next to wealth, n, and the parameters, so the snapshot `state()` returns is sufficient to continue the betting test-martingale. `eProcess({ resume })` rebuilds a process from such a snapshot, and `sequentialPairedGate({ resume })` rebuilds the gate's observe-stream from the new exported `SequentialStreamState` (the e-process state plus the gate decision). A process or gate interrupted at any n and resumed from a JSON round-trip of its state produces the identical wealth sequence, observation sequence, decision, and final state as an uninterrupted run. The snapshot never supplies the parameters: a snapshot recorded under a different alpha, maxBet, null boundary, or threshold is refused with a `ValidationError`, as is one whose fields cannot all be true at once (running sums out of range, a latch without its n, a gate decision this configuration could not have reached). `sequentialDecide` needs no snapshot: it is replayed from the generation history it is handed.
|
|
121
|
+
|
|
122
|
+
---
|
|
123
|
+
|
|
124
|
+
## [0.153.0] — 2026-08-21
|
|
125
|
+
|
|
126
|
+
### Changed
|
|
127
|
+
|
|
128
|
+
- One canonical-JSON encoder (#646, part 1 of 3). Every non-durable digest and stable serialization in the package now delegates to `ledger-core/canonical` (RFC 8785): `canonicalJson`/`contentHash` (the verdict-cache pair the attestation, campaign-manifest, coverage, and trace-repair digests build on), `hashScenarios`/`Dataset.toJsonl`, `serializeFeedbackTrajectoriesJsonl`, the `runControl` default state/action fingerprints, the code-agent-session `configHash`, `canonicalDigest` (loop provenance), and `argHash`. For plain JSON data the bytes — and therefore every persisted digest over plain data — do not change. What changes: a value with no faithful canonical form is refused with `LedgerCanonicalizationError` instead of coerced. `Date`/`toJSON` objects no longer serialize (pass ISO strings), an `undefined`-valued field no longer silently drops or collides with the absent field, and integer-like object keys sort as strings where the sort-into-a-new-object encoders emitted them in numeric enumeration order first.
|
|
129
|
+
- The verdict cache key scheme is `v2:` + the content hash (`VERDICT_CACHE_KEY_SCHEME`). Entries a store holds under the unprefixed scheme miss once and repopulate; no verdict is wrong afterward, only cold.
|
|
130
|
+
- `runControl`'s default fingerprint no longer falls back to `String(value)` for an unserializable state or action — two states that differ in such a field must not fingerprint alike, so it throws; callers with such state supply `stopPolicies.stateFingerprint`/`actionFingerprint`. `argHash` likewise refuses args with no canonical form; uncaptured (`undefined`) args still key to `'undefined'`.
|
|
131
|
+
|
|
132
|
+
---
|
|
133
|
+
|
|
134
|
+
## [0.152.0] — 2026-08-21
|
|
135
|
+
|
|
136
|
+
### Added
|
|
137
|
+
|
|
138
|
+
- `runCampaign({ cellRetry })` — opt-in bounded in-run retry of failed cells. A cell whose failure the caller's `retryable` predicate accepts is dispatched again in the same slot (same `cellId`, same seed, same cost tags) until it succeeds or `attempts` is exhausted, so one router 503 no longer leaves campaign coverage incomplete and forces `runImprovementLoop` to refuse the holdout comparison. Every attempt charges the shared cost ledger (`costUsd` and `costCallIds` on the final cell cover all attempts); a retried attempt keeps its evidence at `<cell>/failure-receipt.attempt-<n>.json` while a final failure keeps `failure-receipt.json`; the final cell records `retryAttempts`; `abortOnCellError` fires only after the last attempt fails, and a cancelled campaign never retries. The exported `transientDispatchFailure()` predicate retries only dispatch-stage failures that `isTransientTransportFailure` classifies as infrastructure hiccups — a judge-stage failure is never transport. `selfImprove({ cellRetry })` forwards the policy to baseline, candidate, and holdout campaigns. Fail-closed default: no retry unless opted in.
|
|
139
|
+
|
|
140
|
+
---
|
|
141
|
+
|
|
142
|
+
## [0.151.0] — 2026-08-20
|
|
143
|
+
|
|
144
|
+
### Added
|
|
145
|
+
|
|
146
|
+
- `selectParent` on `runOptimization()` and `selfImprove()` (#632): a policy for which scored surface each generation MUTATES. Absent, the loop stays incumbent-anchored and the recorded `parentSurfaceHash` lineage is a chain. Present, the selector receives the Pareto frontier so far, the measured incumbent, the generation history, and the generation index, and returns one frontier parent; the loop hands it to `propose()` as `currentSurface` plus the new `ProposeContext.parentOutcome`, records it as every candidate's `parentSurfaceHash` / `parentComposite` / `observedDeltaFromParent`, and refuses a parent the run never measured to completion. Promotion is unchanged: a candidate still has to beat the incumbent, and `incumbentOutcome` stays the global bar. `crowdedFrontierParent({ seed })` from `/campaign` is the provided policy, a seeded NSGA-II crowded tournament over `paretoFrontierWithCrowding` that prefers isolated frontier parents and is deterministic per `(seed, generation)`.
|
|
147
|
+
- Two evidence records for the agent-engine integration arc. `agent-engine-shim-live-proof` (MEASURED-ONCE): the Anthropic loopback shim serves an unmodified Claude Code CLI driving a GEPA autoresearch optimization on the published 0.150.2 npm + PyPI packages, fully metered — 8/8 anthropic-wire requests admitted and completed, totalCostUsd 0.029778336 with accountingComplete true, baseline 0.286 -> winner 1.0 on the deterministic toy objective. `gepa-bridge-machinery-certification` (MEASURED-ONCE, negative disclosed): the GEPA bridge machinery is certified on AIME-2025 (150/150 evaluations, 22 proposals, measured-worse candidates rejected, dual-count integrity held), while the lift verdict is NOT measured — two held-out items exceed glm-5.3's 16000-token reasoning envelope / 480s dispatch wall, and the fail-loud comparison refuses partial verdicts by design.
|
|
148
|
+
|
|
149
|
+
---
|
|
150
|
+
|
|
151
|
+
## [0.150.2] — 2026-08-20
|
|
152
|
+
|
|
153
|
+
### Added
|
|
154
|
+
|
|
155
|
+
- `economics.spend` on the supervisor-run report and `spendUsd` on the rollup (#660): both total-spend measurements as named fields with their denominators, never one bare number. `spend.journalDerived` (journal `metered` + `settled` rows) answers execution accounting — what execution observably consumed; `spend.closeRecord` (loops `state.json` `result.spentUsd`, or Runtime `result.json` `spentTotal.usd`, which the analyzer now reads) answers billing — what the store recorded as settled at close. Each run-level measurement carries its record count; each rollup measurement carries `runs`, its own denominator, because the two sums cover different run sets (measured in discovery-lab: 4.762B input tokens journal-derived over 918 runs vs 4.780B close-record over 868 runs, ~0.4% apart). Divergence is a signal to read, not an error. `totalUsd` stays as the collapsed compatibility field; its docstring names the pick order and points to `spend`.
|
|
156
|
+
- `summarizeNumberSeries(values)` on the statistics surface (#659): the distribution fold (`n` / `min` / `p50` / `p90` / `max` / `sum`, nearest-rank quantiles) that previously lived only inside the supervisor-run analyzer as the worker-wall summary. Returns the exported `SeriesDistribution`, or `null` for an empty series. `WallDistribution` is now a type alias of `SeriesDistribution` and the analyzer reuses the helper in place, so the two folds cannot drift.
|
|
157
|
+
|
|
158
|
+
### Fixed
|
|
159
|
+
|
|
160
|
+
- The `/v1/messages` shim translation accepts `system`-role messages inside the `messages` array. Measured blocker (live proof against published 0.150.1): claude CLI 2.1.232 injects system-role turns mid-conversation — the `--max-budget-usd` status line ("USD budget: $0/$1; $1 remaining", sent on every engine run because the bridge always passes the budget flag), session-front listings, and task-tool reminders — and the translation refused each with a 400 `message role must be 'user' or 'assistant'`. The canonical contract already allows the system role at any position, so an injected system turn now translates to a canonical system message in place, with the same text-block joining and `cache_control` stripping as the top-level `system` field, which keeps its slot at the front. Two verbatim captured wire bodies ship as test fixtures.
|
|
161
|
+
- `orchestration.supervisorWallMs` now populates on Runtime's file-backed layout (#658). Runtime writes no completion stamp, so the field read `unavailable` on 918 of 918 analyzable discovery-lab runs. When the completion stamp is missing, the analyzer derives the wall from the journal's first-to-last event stamps (spawned / settled / cancelled / metered) and reports the derivation in the new `supervisorWallSource: 'stamps' | 'journal-span'` field — never a silent substitution. `journal-span` is a lower bound; `idleMs`, `idlePct`, and `workerUtilization` integrate to the same bound. A run with explicit stamps reports `stamps` and is unchanged; a run with no journal keeps both fields `unavailable` with the same reason. An inverted stamp pair stays unavailable (corruption, not absence).
|
|
162
|
+
- `readRuntimeSupervisorRun` no longer throws on a run dir without `spawn-journal.jsonl` (#657). It returns the same absent-shaped sources `readLoopsSupervisorRun` returns for a missing store — `journal` and `workers` null, each with a reason — so every dependent metric reads `unavailable`, never 0. Measured in discovery-lab: 294 of 1,212 real run dirs have no spawn journal, and every fleet consumer wrapped the reader in its own guard. `RuntimeReaderOptions.strict: true` opts back into the throw. A journal that exists but cannot be parsed still throws. `SupervisorRunSources` gains optional `journalMissingReason`, which the analyzer uses verbatim so a non-loops layout names its own journal file.
|
|
163
|
+
- Docs freshness sweep against the 0.150.1 surface. `docs/campaign-proposers.md`: the metered agent CLI path records its measured status (tool calls translate, dual-wire receipts must match admitted attempts), names the `-inf` trap (an agent engine's one registering evaluation costs the full train set against `maxEvaluations`), and the Runtime Knobs table gains `maxEvaluations` (agent engines), `budget.maxRequests` (agent engines), and `expectUsage`; the unproxied-engine example is labeled as the unproxied path. `docs/concepts.md` no longer claims three exported bias probes: `verbosityBias` is the one shipped probe, and `JudgeInsight.positionalBias`/`selfPreference` are caller-filled fields. `docs/adapters-observability.md` drops the deleted `createOtelBridge` design note (the module left in the stranded-module deletions) and points at `createHostedClient` + `/v1/traces/ingest` instead. `clients/python/README.md` documents the fully metered agent-engine path behind `optimizer.anthropicEndpoint`. `docs/wire-protocol.md` version example matches the current release.
|
|
164
|
+
- `examples/agent-engine-optimizer/`: the first example of the metered agent-engine path. `selfImprove` + `gepaOptimizationMethod` with the `autoresearch` engine drives a real `claude` CLI through the loopback Anthropic route (`optimizer.anthropicEndpoint: true`). The example encodes the measured constraints as code: `maxEvaluations` at least the train-set size (one registering aggregate eval costs the whole training pool, enforced at startup), `expectUsage: 'off'` for a deterministic no-LLM evaluator, and output-token headroom for reasoning models. The README documents the receipt fields the run prints and the 0.150.2 system-role translation prerequisite for unmodified CLI runs.
|
|
165
|
+
- `runDispatchServer` no longer aborts every dispatch on current Node. The server keyed client-disconnect detection on the request stream's `close` event, which Node fires when the request BODY completes, so every worker call aborted ~immediately and the two-process `distributed-driver` example failed end to end. Disconnect detection now keys on the response stream's `close` with `writableEnded` still false. `src/adapters/http.test.ts` covers both directions: a slow dispatch completes untouched, and a mid-flight client abort reaches the worker's `ctx.signal`.
|
|
166
|
+
- Examples freshness sweep — every directory under `examples/` re-verified compile + run as documented:
|
|
167
|
+
- `distributed-driver` names its dispatch (`dispatchRef`) — `httpDispatch` returns an anonymous function, which the campaign manifest rejects — and sets `expectUsage: 'off'` for its stub worker.
|
|
168
|
+
- `multi-shot-optimization` and `selfimprove-quickstart` printed `hold` while their READMEs promised `ship`: the gate's exact paired test cannot reach 95% significance below six paired holdout observations. Both now hold six cases and ship; the READMEs state the floor.
|
|
169
|
+
- `fine-tune-with-prime-rl`: the shipped fixture predated the 0.126 `RunRecord` shape (`costProvenance`, `terminalOutcome`) and its rows sat on the holdout split, so the documented command crashed and, once past validation, exported ZERO rows silently. The fixture is regenerated on the `search` split with top-level `prompt`/`completion` (`outcome.raw` is numeric-only by contract), text lookups throw on missing fields instead of exporting placeholder rows, an empty export now fails loudly, and the README command paths and pinned output are verified.
|
|
170
|
+
- `customer-otel-traces` read the retired `failureMode` field, so its "Failures" section silently vanished; it now counts `outcome.raw.error_span_count` and the README pins the real report, including the failure-class recommendation.
|
|
171
|
+
- `hosted-ingest-server` gains the README the index always pointed at, and the file-header `curl` now carries the required `X-Tangle-Wire-Version` header (verified against the live server: without it, 400).
|
|
172
|
+
- `same-sandbox-harness`'s documented `tsx -e` snippet used top-level await, which tsx rejects in string-eval; the README now uses a dynamic import (verified to run).
|
|
173
|
+
- `foreign-agent-quickstart` moved to the standard `LLM_BASE_URL`/`LLM_API_KEY`/`LLM_MODEL` variables and declares `expectUsage: 'off'` for its unmetered demo transport.
|
|
174
|
+
- Optimizer examples (`self-improve-optimizer`, `compare-optimization-methods`, GSM8K, AppWorld) encode the measured GEPA lessons: `LLM_BASE_URL` is required (no silent `api.openai.com` default), every GEPA engine run pins `reflection_lm_kwargs: { num_retries: 0 }` via the shared `examples/_shared/gepa-reflection.ts`, worker `maxTokens` is env-tunable with a reasoning-model headroom note, and `GEPA_MAX_EVALUATIONS` documents the >= train-partition-size floor.
|
|
175
|
+
- Model ids refreshed across examples: retired `gpt-4o`/`gpt-4.1-mini`/`claude-sonnet-4-6` literals replaced with served ids (`deepseek-v4-flash`, `glm-5.3`; RunRecord literals carry the required snapshot suffix).
|
|
176
|
+
- Removed `examples/benchmarks/appworld/halo_chat.py` and `halo-chat.sh`: launchers for an external analysis engine that no documented example references. `run-bench.ts` requires `APPWORLD_DIR` explicitly instead of defaulting to a machine-local path.
|
|
8
177
|
|
|
9
178
|
---
|
|
10
179
|
|
|
@@ -1179,6 +1348,10 @@ threshold and should be re-run on this release.
|
|
|
1179
1348
|
|
|
1180
1349
|
## [0.130.1] - 2026-07-26 - safe DSPy disk caching
|
|
1181
1350
|
|
|
1351
|
+
### Changed
|
|
1352
|
+
|
|
1353
|
+
- Doc comments on the public campaign search APIs: `FileSearchLedger`, the three search-ledger error classes, `canonicalDigest` in provenance, and the surface-identity helpers (`87c02e6`, #444).
|
|
1354
|
+
|
|
1182
1355
|
### Fixed
|
|
1183
1356
|
|
|
1184
1357
|
- `DspyJudgeMetric` now rejects DSPy's unrestricted disk-cache pickle mode at construction and on every metric call.
|
|
@@ -1186,6 +1359,10 @@ threshold and should be re-run on this release.
|
|
|
1186
1359
|
|
|
1187
1360
|
## [0.130.0] - 2026-07-26 - current dependency and build cohort
|
|
1188
1361
|
|
|
1362
|
+
### Added
|
|
1363
|
+
|
|
1364
|
+
- `./ledger-core` subpath: the generic journal machinery extracted from `search-ledger` (`25ce61b`, #441). It holds the canonical-JSON SHA-256 hash chain, idempotent append by `eventId`, chain verification, replay to projection, the per-path mutex with an atomic lock file, and the fsync discipline, with no campaign vocabulary. `search-ledger` is now the campaign codec bound to `FileLedgerJournal`; files, error classes, messages, and check order are unchanged.
|
|
1365
|
+
|
|
1189
1366
|
### Changed
|
|
1190
1367
|
|
|
1191
1368
|
- Updated Agent Core to `0.4.24` and Agent Interface to `0.35.0`.
|
package/README.md
CHANGED
|
@@ -110,21 +110,25 @@ Every row is a function you call. Each links to a runnable example.
|
|
|
110
110
|
## Configure Model Calls
|
|
111
111
|
|
|
112
112
|
Benchmarks, user drivers, executors, built-in judges, completion checkers, and judge adapters all take the same `ChatClient`.
|
|
113
|
+
You own model execution: Agent Eval issues no provider request and never receives a provider credential.
|
|
113
114
|
|
|
114
115
|
```ts
|
|
115
116
|
import { createChatClient } from '@tangle-network/agent-eval'
|
|
116
117
|
|
|
117
118
|
const chat = createChatClient({
|
|
118
|
-
transport: '
|
|
119
|
-
apiKey: process.env.TANGLE_API_KEY!,
|
|
119
|
+
transport: 'custom',
|
|
120
120
|
defaultModel: 'openai/gpt-4.1',
|
|
121
121
|
maximumAttempts: 3,
|
|
122
|
+
chat: async (request, opts) => myProviderClient(request, opts),
|
|
122
123
|
})
|
|
123
124
|
```
|
|
124
125
|
|
|
125
|
-
|
|
126
|
+
On Agent Runtime, `profileChatClient({ profile, executor, context })` from `@tangle-network/agent-runtime/kernel` is that transport: every call runs one exact `AgentProfile` and reports its measured usage, retries, and served model identity.
|
|
127
|
+
Use `sandbox-sdk` for Sandbox and `mock` in tests.
|
|
126
128
|
A custom adapter must return a `ChatResponse` and declare `maximumAttempts` before a capped cost account can dispatch it.
|
|
127
129
|
|
|
130
|
+
`ChatResponse` carries the whole execution record across that boundary: the served model id, measured input/output/reasoning/cached tokens, billed USD or an explicit unknown, the finish reason, and the per-token log probabilities the expectation judge scores on.
|
|
131
|
+
|
|
128
132
|
The official GEPA and SkillOpt optimizers run through a Python bridge.
|
|
129
133
|
Install commands, version pins, and the reason for each pin:
|
|
130
134
|
[GEPA](./docs/campaign-proposers.md#install-official-gepa),
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { i as makeRng } from "./internal-BDHPCnjk.js";
|
|
1
2
|
import { n as observedScore } from "./reward-nw2xZGZG.js";
|
|
2
3
|
//#region src/rl/active-curriculum.ts
|
|
3
4
|
/**
|
|
@@ -103,7 +104,7 @@ function thompsonCurriculum(observations, candidateCells, opts) {
|
|
|
103
104
|
const threshold = opts.decisionThreshold ?? .5;
|
|
104
105
|
const alpha0 = opts.priorAlpha ?? 1;
|
|
105
106
|
const beta0 = opts.priorBeta ?? 1;
|
|
106
|
-
const rng = makeRng(opts.seed);
|
|
107
|
+
const rng = makeRng(opts.seed, observations.map((observation) => observation.score));
|
|
107
108
|
const grouped = /* @__PURE__ */ new Map();
|
|
108
109
|
for (const o of observations) {
|
|
109
110
|
const k = `${o.variantId}::${o.scenarioId}`;
|
|
@@ -168,17 +169,6 @@ function observationsFromRunRecords(runs, opts = {}) {
|
|
|
168
169
|
}
|
|
169
170
|
return out;
|
|
170
171
|
}
|
|
171
|
-
function makeRng(seed) {
|
|
172
|
-
if (seed === void 0) return Math.random;
|
|
173
|
-
let s = seed >>> 0;
|
|
174
|
-
return () => {
|
|
175
|
-
s = s + 1831565813 >>> 0;
|
|
176
|
-
let t = s;
|
|
177
|
-
t = Math.imul(t ^ t >>> 15, t | 1);
|
|
178
|
-
t ^= t + Math.imul(t ^ t >>> 7, t | 61);
|
|
179
|
-
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
180
|
-
};
|
|
181
|
-
}
|
|
182
172
|
/**
|
|
183
173
|
* Sample from Beta(α, β) via the Marsaglia–Tsang method using two Gamma
|
|
184
174
|
* variates. Accuracy is good for α, β > 1; we floor the parameters at 1
|
|
@@ -211,4 +201,4 @@ function sampleGamma(shape, rng) {
|
|
|
211
201
|
//#endregion
|
|
212
202
|
export { thompsonCurriculum as n, varianceBasedCurriculum as r, observationsFromRunRecords as t };
|
|
213
203
|
|
|
214
|
-
//# sourceMappingURL=active-curriculum-
|
|
204
|
+
//# sourceMappingURL=active-curriculum-CD5TU2yW.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"active-curriculum-CD5TU2yW.js","names":[],"sources":["../src/rl/active-curriculum.ts"],"sourcesContent":["/**\n * Adaptive curriculum / active scenario selection.\n *\n * Fixed scenario sets waste sample budget on cells the policy already\n * passes (no information left) and cells the policy never passes (no\n * gradient available either). Active learning over scenarios fixes this\n * by allocating the next sample budget to cells where the policy's\n * outcome is *uncertain* — those carry the most decision-relevant signal.\n *\n * This module ships two complementary strategies:\n *\n * 1. **Variance-based** — score each (variant, scenario) cell by the\n * empirical variance of past observations. Allocate next-round budget\n * proportional to variance. Standard active-learning-by-uncertainty\n * heuristic; works well when the policy is non-deterministic and\n * cells differ in observation noise.\n *\n * 2. **Bandit-based (Thompson sampling)** — model each (variant,\n * scenario) cell as a Beta-Bernoulli arm; sample a posterior; pick\n * cells whose posterior mean is closest to the per-scenario decision\n * threshold. The right primitive when scenarios are\n * \"pass/fail\" rather than continuous, and when promotion gates fire\n * at a known threshold (e.g., 0.5).\n *\n * The output is a *next-round budget allocation* — a list of (variant,\n * scenario, count) triples. The consumer's matrix runner consumes the\n * allocation, runs those cells, feeds the new observations back. Loop.\n *\n * Out of scope (deliberate): scenario *generation* — that's the\n * adversarial primitive's job. This module allocates over an existing\n * scenario pool.\n */\n\nimport { observedScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\nimport { makeRng } from '../statistics/internal'\n\nexport interface CellObservation {\n variantId: string\n scenarioId: string\n /** Observed score in [0, 1]. */\n score: number\n /** For Bernoulli arms — derive from the score with a threshold if needed. */\n pass?: boolean\n}\n\nexport interface CurriculumAllocation {\n variantId: string\n scenarioId: string\n /** How many additional reps to run on this cell. */\n count: number\n /** Strategy-specific reason for the allocation. */\n reason: string\n}\n\nexport interface VarianceCurriculumOptions {\n /** Total reps to allocate across all cells. */\n budget: number\n /**\n * Smoothing prior on variance — keeps the allocator from concentrating\n * on a cell with one observation just because its 1-sample variance is\n * 0. Default 0.05.\n */\n variancePrior?: number\n /**\n * Minimum reps per cell — even when the variance estimate is low, give\n * every cell at least this many. Default 1.\n */\n floorPerCell?: number\n}\n\n/**\n * Variance-proportional allocation. For each cell, estimate variance from\n * past observations + a prior, then allocate the budget proportional to\n * (sqrt(variance) + 1/sqrt(n)) — a classical optimal-allocation rule\n * (Neyman 1934) that balances \"explore noisy cells\" with \"explore\n * under-sampled cells.\"\n */\nexport function varianceBasedCurriculum(\n observations: CellObservation[],\n candidateCells: Array<{ variantId: string; scenarioId: string }>,\n opts: VarianceCurriculumOptions,\n): CurriculumAllocation[] {\n const variancePrior = opts.variancePrior ?? 0.05\n const floor = opts.floorPerCell ?? 1\n const budget = opts.budget\n\n const grouped = new Map<string, number[]>()\n for (const o of observations) {\n const k = `${o.variantId}::${o.scenarioId}`\n const arr = grouped.get(k) ?? []\n arr.push(o.score)\n grouped.set(k, arr)\n }\n\n const cellStats = candidateCells.map((c) => {\n const k = `${c.variantId}::${c.scenarioId}`\n const samples = grouped.get(k) ?? []\n const n = samples.length\n const mean = n === 0 ? 0.5 : samples.reduce((s, v) => s + v, 0) / n\n const variance =\n n < 2\n ? variancePrior\n : samples.reduce((s, v) => s + (v - mean) ** 2, 0) / (n - 1) + variancePrior\n // Neyman optimal allocation: weight ∝ √variance; add √(1/n) to break\n // ties toward under-sampled cells.\n const weight = Math.sqrt(variance) + 1 / Math.sqrt(Math.max(1, n))\n return { variantId: c.variantId, scenarioId: c.scenarioId, n, mean, variance, weight }\n })\n\n // Reserve floor*N for the floor; allocate the rest proportional to weight.\n const floorTotal = floor * cellStats.length\n if (floorTotal >= budget) {\n const each = Math.max(1, Math.floor(budget / Math.max(1, cellStats.length)))\n return cellStats.map((c) => ({\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: each,\n reason: `floor allocation (budget tight; n=${c.n})`,\n }))\n }\n const remaining = budget - floorTotal\n const totalWeight = cellStats.reduce((s, c) => s + c.weight, 0)\n return cellStats.map((c) => {\n const proportional = totalWeight === 0 ? 0 : Math.round((c.weight / totalWeight) * remaining)\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: floor + proportional,\n reason: `variance ${c.variance.toFixed(3)} (n=${c.n}, mean=${c.mean.toFixed(3)})`,\n }\n })\n}\n\nexport interface ThompsonCurriculumOptions {\n budget: number\n /**\n * The per-scenario decision threshold. Cells whose posterior mean is\n * closest to this get the most budget — that's where the next observation\n * has the highest information value for the gate decision. Default 0.5.\n */\n decisionThreshold?: number\n /** Beta prior parameters. Default α=β=1 (uniform). */\n priorAlpha?: number\n priorBeta?: number\n /** Seed the Thompson sampler. Absent, the seed is derived from the observed\n * scores, so the same observations reproduce the same allocation. */\n seed?: number\n}\n\n/**\n * Thompson-sampling-style allocation for pass/fail cells. For each cell:\n *\n * - Maintain Beta(α + passes, β + failures) posterior on pass-rate\n * - Allocation weight ∝ exp(-((sampledMean - threshold) / σ)^2):\n * cells whose sampled posterior straddles the decision boundary get\n * the most weight; cells already clearly above or below get less.\n *\n * This is the right primitive when promotion gates fire at a known\n * threshold and you want to sharpen the posterior near the boundary.\n */\nexport function thompsonCurriculum(\n observations: CellObservation[],\n candidateCells: Array<{ variantId: string; scenarioId: string }>,\n opts: ThompsonCurriculumOptions,\n): CurriculumAllocation[] {\n const threshold = opts.decisionThreshold ?? 0.5\n const alpha0 = opts.priorAlpha ?? 1\n const beta0 = opts.priorBeta ?? 1\n const rng = makeRng(\n opts.seed,\n observations.map((observation) => observation.score),\n )\n\n const grouped = new Map<string, { passes: number; failures: number }>()\n for (const o of observations) {\n const k = `${o.variantId}::${o.scenarioId}`\n const cur = grouped.get(k) ?? { passes: 0, failures: 0 }\n const pass = o.pass ?? o.score >= threshold\n if (pass) cur.passes += 1\n else cur.failures += 1\n grouped.set(k, cur)\n }\n\n const stats = candidateCells.map((c) => {\n const k = `${c.variantId}::${c.scenarioId}`\n const cur = grouped.get(k) ?? { passes: 0, failures: 0 }\n const a = alpha0 + cur.passes\n const b = beta0 + cur.failures\n // Sample a single Beta draw — the Thompson signal.\n const sampled = sampleBeta(a, b, rng)\n const distance = Math.abs(sampled - threshold)\n // Information-near-threshold weight: closer = higher.\n // Use Gaussian-shaped kernel with σ tuned to posterior std.\n const variance = (a * b) / ((a + b) ** 2 * (a + b + 1))\n const sigma = Math.max(0.05, Math.sqrt(variance))\n const weight = Math.exp(-((distance / sigma) ** 2))\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n n: cur.passes + cur.failures,\n sampled,\n sigma,\n weight,\n a,\n b,\n }\n })\n\n const totalWeight = stats.reduce((s, c) => s + c.weight, 0)\n return stats.map((c) => {\n const proportional = totalWeight === 0 ? 0 : Math.round((c.weight / totalWeight) * opts.budget)\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: Math.max(0, proportional),\n reason: `Beta(${c.a.toFixed(1)},${c.b.toFixed(1)}) sample=${c.sampled.toFixed(3)} (target ${threshold})`,\n }\n })\n}\n\n/** Convenience: extract `CellObservation[]` directly from `RunRecord[]`. */\nexport function observationsFromRunRecords(\n runs: RunRecord[],\n opts: { passThreshold?: number; useHoldout?: boolean } = {},\n): CellObservation[] {\n const threshold = opts.passThreshold ?? 0.5\n const useHoldout = opts.useHoldout ?? true\n const out: CellObservation[] = []\n for (const r of runs) {\n if (!r.scenarioId) continue\n // Ungated on purpose, and the precedence is caller policy, not a default:\n // `useHoldout: false` means \"score this curriculum on the search split when\n // both exist\". This feeds sampling COUNTS, not an exported reward. Known\n // risk: a gamed run's high score inflates the cell's Beta posterior, so the\n // curriculum stops sampling a cell it wrongly believes is solved. The fix\n // for that is an upstream filter on gated records — zeroing the score here\n // would push the posterior the opposite way and be equally wrong.\n const score = observedScore(r, useHoldout ? 'holdout' : 'search')\n if (typeof score !== 'number' || !Number.isFinite(score)) continue\n out.push({\n variantId: r.candidateId,\n scenarioId: r.scenarioId,\n score,\n pass: score >= threshold,\n })\n }\n return out\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\n/**\n * Sample from Beta(α, β) via the Marsaglia–Tsang method using two Gamma\n * variates. Accuracy is good for α, β > 1; we floor the parameters at 1\n * to avoid degenerate cases.\n */\nfunction sampleBeta(alpha: number, beta: number, rng: () => number): number {\n const a = Math.max(1, alpha)\n const b = Math.max(1, beta)\n const x = sampleGamma(a, rng)\n const y = sampleGamma(b, rng)\n return x / (x + y)\n}\n\nfunction sampleGamma(shape: number, rng: () => number): number {\n // Marsaglia–Tsang for shape ≥ 1.\n const d = shape - 1 / 3\n const c = 1 / Math.sqrt(9 * d)\n while (true) {\n let x: number\n let v: number\n do {\n const u1 = rng() || 1e-12\n const u2 = rng() || 1e-12\n // Box-Muller for a normal sample.\n x = Math.sqrt(-2 * Math.log(u1)) * Math.cos(2 * Math.PI * u2)\n v = 1 + c * x\n } while (v <= 0)\n v = v * v * v\n const u = rng()\n if (u < 1 - 0.0331 * x ** 4) return d * v\n if (Math.log(u) < 0.5 * x * x + d * (1 - v + Math.log(v))) return d * v\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA8EA,SAAgB,wBACd,cACA,gBACA,MACwB;CACxB,MAAM,gBAAgB,KAAK,iBAAiB;CAC5C,MAAM,QAAQ,KAAK,gBAAgB;CACnC,MAAM,SAAS,KAAK;CAEpB,MAAM,0BAAU,IAAI,IAAsB;CAC1C,KAAK,MAAM,KAAK,cAAc;EAC5B,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,MAAM,QAAQ,IAAI,CAAC,KAAK,CAAC;EAC/B,IAAI,KAAK,EAAE,KAAK;EAChB,QAAQ,IAAI,GAAG,GAAG;CACpB;CAEA,MAAM,YAAY,eAAe,KAAK,MAAM;EAC1C,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,UAAU,QAAQ,IAAI,CAAC,KAAK,CAAC;EACnC,MAAM,IAAI,QAAQ;EAClB,MAAM,OAAO,MAAM,IAAI,KAAM,QAAQ,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;EAClE,MAAM,WACJ,IAAI,IACA,gBACA,QAAQ,QAAQ,GAAG,MAAM,KAAK,IAAI,SAAS,GAAG,CAAC,KAAK,IAAI,KAAK;EAGnE,MAAM,SAAS,KAAK,KAAK,QAAQ,IAAI,IAAI,KAAK,KAAK,KAAK,IAAI,GAAG,CAAC,CAAC;EACjE,OAAO;GAAE,WAAW,EAAE;GAAW,YAAY,EAAE;GAAY;GAAG;GAAM;GAAU;EAAO;CACvF,CAAC;CAGD,MAAM,aAAa,QAAQ,UAAU;CACrC,IAAI,cAAc,QAAQ;EACxB,MAAM,OAAO,KAAK,IAAI,GAAG,KAAK,MAAM,SAAS,KAAK,IAAI,GAAG,UAAU,MAAM,CAAC,CAAC;EAC3E,OAAO,UAAU,KAAK,OAAO;GAC3B,WAAW,EAAE;GACb,YAAY,EAAE;GACd,OAAO;GACP,QAAQ,qCAAqC,EAAE,EAAE;EACnD,EAAE;CACJ;CACA,MAAM,YAAY,SAAS;CAC3B,MAAM,cAAc,UAAU,QAAQ,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;CAC9D,OAAO,UAAU,KAAK,MAAM;EAC1B,MAAM,eAAe,gBAAgB,IAAI,IAAI,KAAK,MAAO,EAAE,SAAS,cAAe,SAAS;EAC5F,OAAO;GACL,WAAW,EAAE;GACb,YAAY,EAAE;GACd,OAAO,QAAQ;GACf,QAAQ,YAAY,EAAE,SAAS,QAAQ,CAAC,EAAE,MAAM,EAAE,EAAE,SAAS,EAAE,KAAK,QAAQ,CAAC,EAAE;EACjF;CACF,CAAC;AACH;;;;;;;;;;;;AA6BA,SAAgB,mBACd,cACA,gBACA,MACwB;CACxB,MAAM,YAAY,KAAK,qBAAqB;CAC5C,MAAM,SAAS,KAAK,cAAc;CAClC,MAAM,QAAQ,KAAK,aAAa;CAChC,MAAM,MAAM,QACV,KAAK,MACL,aAAa,KAAK,gBAAgB,YAAY,KAAK,CACrD;CAEA,MAAM,0BAAU,IAAI,IAAkD;CACtE,KAAK,MAAM,KAAK,cAAc;EAC5B,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,MAAM,QAAQ,IAAI,CAAC,KAAK;GAAE,QAAQ;GAAG,UAAU;EAAE;EAEvD,IADa,EAAE,QAAQ,EAAE,SAAS,WACxB,IAAI,UAAU;OACnB,IAAI,YAAY;EACrB,QAAQ,IAAI,GAAG,GAAG;CACpB;CAEA,MAAM,QAAQ,eAAe,KAAK,MAAM;EACtC,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,MAAM,QAAQ,IAAI,CAAC,KAAK;GAAE,QAAQ;GAAG,UAAU;EAAE;EACvD,MAAM,IAAI,SAAS,IAAI;EACvB,MAAM,IAAI,QAAQ,IAAI;EAEtB,MAAM,UAAU,WAAW,GAAG,GAAG,GAAG;EACpC,MAAM,WAAW,KAAK,IAAI,UAAU,SAAS;EAG7C,MAAM,WAAY,IAAI,MAAO,IAAI,MAAM,KAAK,IAAI,IAAI;EACpD,MAAM,QAAQ,KAAK,IAAI,KAAM,KAAK,KAAK,QAAQ,CAAC;EAChD,MAAM,SAAS,KAAK,IAAI,GAAG,WAAW,UAAU,EAAE;EAClD,OAAO;GACL,WAAW,EAAE;GACb,YAAY,EAAE;GACd,GAAG,IAAI,SAAS,IAAI;GACpB;GACA;GACA;GACA;GACA;EACF;CACF,CAAC;CAED,MAAM,cAAc,MAAM,QAAQ,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;CAC1D,OAAO,MAAM,KAAK,MAAM;EACtB,MAAM,eAAe,gBAAgB,IAAI,IAAI,KAAK,MAAO,EAAE,SAAS,cAAe,KAAK,MAAM;EAC9F,OAAO;GACL,WAAW,EAAE;GACb,YAAY,EAAE;GACd,OAAO,KAAK,IAAI,GAAG,YAAY;GAC/B,QAAQ,QAAQ,EAAE,EAAE,QAAQ,CAAC,EAAE,GAAG,EAAE,EAAE,QAAQ,CAAC,EAAE,WAAW,EAAE,QAAQ,QAAQ,CAAC,EAAE,WAAW,UAAU;EACxG;CACF,CAAC;AACH;;AAGA,SAAgB,2BACd,MACA,OAAyD,CAAC,GACvC;CACnB,MAAM,YAAY,KAAK,iBAAiB;CACxC,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,MAAyB,CAAC;CAChC,KAAK,MAAM,KAAK,MAAM;EACpB,IAAI,CAAC,EAAE,YAAY;EAQnB,MAAM,QAAQ,cAAc,GAAG,aAAa,YAAY,QAAQ;EAChE,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;EAC1D,IAAI,KAAK;GACP,WAAW,EAAE;GACb,YAAY,EAAE;GACd;GACA,MAAM,SAAS;EACjB,CAAC;CACH;CACA,OAAO;AACT;;;;;;AASA,SAAS,WAAW,OAAe,MAAc,KAA2B;CAC1E,MAAM,IAAI,KAAK,IAAI,GAAG,KAAK;CAC3B,MAAM,IAAI,KAAK,IAAI,GAAG,IAAI;CAC1B,MAAM,IAAI,YAAY,GAAG,GAAG;CAE5B,OAAO,KAAK,IADF,YAAY,GAAG,GACT;AAClB;AAEA,SAAS,YAAY,OAAe,KAA2B;CAE7D,MAAM,IAAI,QAAQ,IAAI;CACtB,MAAM,IAAI,IAAI,KAAK,KAAK,IAAI,CAAC;CAC7B,OAAO,MAAM;EACX,IAAI;EACJ,IAAI;EACJ,GAAG;GACD,MAAM,KAAK,IAAI,KAAK;GACpB,MAAM,KAAK,IAAI,KAAK;GAEpB,IAAI,KAAK,KAAK,KAAK,KAAK,IAAI,EAAE,CAAC,IAAI,KAAK,IAAI,IAAI,KAAK,KAAK,EAAE;GAC5D,IAAI,IAAI,IAAI;EACd,SAAS,KAAK;EACd,IAAI,IAAI,IAAI;EACZ,MAAM,IAAI,IAAI;EACd,IAAI,IAAI,IAAI,QAAS,KAAK,GAAG,OAAO,IAAI;EACxC,IAAI,KAAK,IAAI,CAAC,IAAI,KAAM,IAAI,IAAI,KAAK,IAAI,IAAI,KAAK,IAAI,CAAC,IAAI,OAAO,IAAI;CACxE;AACF"}
|
|
@@ -50,7 +50,9 @@ declare class AgentProfileCellValidationError extends ValidationError {
|
|
|
50
50
|
declare function buildAgentProfileCell(input: AgentProfileCellInput): Promise<AgentProfileCell>;
|
|
51
51
|
declare function agentProfileCellHashMaterial(cell: AgentProfileCell): Omit<AgentProfileCell, 'cellId'>;
|
|
52
52
|
/**
|
|
53
|
-
* Verify an `AgentProfileCell`'s `cellId` matches the sha256 of its hash-material
|
|
53
|
+
* Verify an `AgentProfileCell`'s `cellId` matches the sha256 of its hash-material
|
|
54
|
+
* fields, confirming the record has not been tampered with. The id names its own
|
|
55
|
+
* digest scheme, so a cell minted by an earlier release verifies under that scheme.
|
|
54
56
|
*/
|
|
55
57
|
declare function verifyAgentProfileCell(cell: AgentProfileCell): Promise<boolean>;
|
|
56
58
|
declare function validateAgentProfileCell(input: unknown): AgentProfileCell;
|
|
@@ -105,4 +107,4 @@ type AgentInterfaceProfileLike = AgentProfile & {
|
|
|
105
107
|
declare function buildAgentInterfaceProfileCell(profile: AgentInterfaceProfileLike, input: Omit<AgentProfileCellInput, 'profileId' | 'sourceProfile'>): Promise<AgentProfileCell>;
|
|
106
108
|
//#endregion
|
|
107
109
|
export { verifyAgentProfileCell as C, validateAgentProfileCell as S, buildAgentInterfaceProfileCell as _, AgentProfileCellSchemaVersion as a, requireAgentProfileCell as b, AgentProfileHarness as c, AgentProfileKind as d, AgentProfileSource as f, assertRunAgentProfileCell as g, agentProfileCellKey as h, AgentProfileCellInput as i, AgentProfileJson as l, agentProfileCellHashMaterial as m, AgentInterfaceProfileLike as n, AgentProfileCellValidationError as o, AgentProfileSourceInput as p, AgentProfileCell as r, AgentProfileDimensionValue as s, AGENT_PROFILE_KINDS as t, AgentProfileJsonObject as u, buildAgentProfileCell as v, toAgentProfileJson as x, groupRunsByAgentProfileCell as y };
|
|
108
|
-
//# sourceMappingURL=agent-profile-cell-
|
|
110
|
+
//# sourceMappingURL=agent-profile-cell-CTOZJUuE.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"agent-profile-cell-CTOZJUuE.d.ts","names":[],"sources":["../src/agent-profile-cell.ts"],"mappings":";;;KAKY;KAEA;GAA4B,cAAc;;KAE1C,sDAKR,qBACA;KAEQ;UAEK;;EAEf;;EAEA;;UAGe;EACf;;EAEA;;EAEA,UAAU;;UAGK;EACf;EACA;EACA;;UAGe;EACf;EACA,eAAe;EACf,UAAU;EACV;EACA;EACA,aAAa,eAAe;;UAGb;EACf,eAAe;EACf;EACA;EACA,eAAe;EACf,UAAU;EACV;EACA;EACA,aAAa,eAAe;;cAGjB,wCAAwC;WAC1C;EACT,YAAY,iBAAiB;;iBAgBT,sBACpB,OAAO,wBACN,QAAQ;iBAMK,6BACd,MAAM,mBACL,KAAK;;;;;;iBAWc,uBAAuB,MAAM,mBAAmB;iBA6BtD,yBAAyB,iBAAiB;iBAqB1C,wBAAwB;EACtC;EACA,eAAe;IACb;iBAUY,oBAAoB;EAClC;EACA,eAAe;;iBAKK,0BAA0B;EAC9C;EACA;EACA;EACA,eAAe;IACb,QAAQ;iBAuBI,4BACd;EAAY;EAAe,eAAe;GAC1C,kBAAkB,MAAM,YAAY;;;;;cA0NzB;;;;WAIX;;KAGU,2BAA2B,kCAAkC;;;;;iBAMzD,mBAAmB,iBAAiB;;KAoBxC,4BAA4B;EAAiB;EAAc;;;;;;;;;;;iBAWjD,+BACpB,SAAS,2BACT,OAAO,KAAK,wDACX,QAAQ"}
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,19 +1,19 @@
|
|
|
1
|
-
import { a as MultiLayerVerifier, c as VerifyOptions
|
|
2
|
-
import {
|
|
1
|
+
import { a as MultiLayerVerifier, c as VerifyOptions } from "../multi-layer-verifier-BUaQ4C17.js";
|
|
2
|
+
import { C as DspyRlmTraceEngineOptions, M as RunScore, N as RunScoreWeights, O as SemanticConceptJudgeInput, S as renderFindingSubject, _ as FindingSubject, a as FAILURE_MODE_KIND_SPEC, b as findingSubjectGrammarPromptFor, c as emitControlIntegrityFindings, d as FindingsStore, f as PersistedFinding, g as FINDING_SUBJECT_SYNTAX, h as FINDING_SUBJECT_KINDS, i as IMPROVEMENT_KIND_SPEC, k as SemanticConceptJudgeOptions, l as DiffPolicy, m as diffFindings, n as KNOWLEDGE_POISONING_KIND_SPEC, o as CONTROL_INTEGRITY_ANALYST, p as defaultIsMaterial, r as KNOWLEDGE_GAP_KIND_SPEC, s as ControlIntegrityAnalyst, t as DEFAULT_TRACE_ANALYST_KINDS, u as FindingsDiff, v as FindingSubjectKind, w as createDspyRlmTraceEngine, x as parseFindingSubject, y as KIND_EXPECTED_SUBJECTS } from "../index-CGtH1piv.js";
|
|
3
3
|
import { b as CustomTokenPricing, c as CostLedgerHandle } from "../cost-ledger-DbQdN3nO.js";
|
|
4
|
-
import { C as TraceEvent, _ as Span, f as Run, n as BudgetLedgerEntry, t as Artifact } from "../schema-
|
|
5
|
-
import {
|
|
6
|
-
import
|
|
7
|
-
import "../
|
|
8
|
-
import { I as TraceAnalystSpan, _ as ProposalFinding, a as AnalystInputKind, b as makeFinding, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as ExecutionProbeRequest, h as ExecutionProbeOutcome, i as AnalystFinding, l as AnalystRunResult, m as ExecutionProbe, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as ProposalFindingOrigin, w as TraceAnalysisStore, x as makeProposalFinding, y as computeFindingId } from "../types-
|
|
9
|
-
import { a as createTraceAnalyst, c as
|
|
10
|
-
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-
|
|
11
|
-
import { a as ExactAnalystBudgetPolicy, c as RegistryRunOpts, i as BudgetPolicy,
|
|
12
|
-
import { C as scoreAnalystFindings, S as traceStoreEvidenceResolver, _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, b as registryBenchmarkRunner, c as AnalystBenchmarkProvenance, d as AnalystBenchmarkSummary, f as AnalystEvidenceExpectation, g as AnalystFindingScore, h as AnalystEvidenceResolver, i as AnalystBenchmarkError, l as AnalystBenchmarkResult, m as AnalystEvidenceResolutionError, n as AnalystBenchmarkDatasetRef, o as AnalystBenchmarkObservation, p as AnalystEvidenceResolution, r as AnalystBenchmarkDescriptor, s as AnalystBenchmarkOutput, t as AnalystBenchmarkCase, u as AnalystBenchmarkRunner, v as AnalystLatencyDistribution, x as runAnalystBenchmark, y as RunAnalystBenchmarkOptions } from "../benchmark-
|
|
13
|
-
import { f as ExternalOptimizerModelExecutionObservation, h as ExternalOptimizerRunnerCommand, l as ExternalOptimizerModelCall } from "../external-optimizer-contracts-
|
|
4
|
+
import { C as TraceEvent, _ as Span, f as Run, n as BudgetLedgerEntry, t as Artifact } from "../schema-Bjgdsn73.js";
|
|
5
|
+
import { s as TraceStore } from "../store-B06JdC56.js";
|
|
6
|
+
import "../index-vrJugRal.js";
|
|
7
|
+
import { _ as CreateChatClientOpts, a as JudgeInput, b as SandboxSdkTransportOpts, f as ChatCallOpts, g as ChatTransport, h as ChatResponse, i as JudgeFn, m as ChatRequest, p as ChatClient, v as CustomTransportOpts, x as createChatClient, y as MockTransportOpts } from "../types-Bfk0uxRj.js";
|
|
8
|
+
import { I as TraceAnalystSpan, _ as ProposalFinding, a as AnalystInputKind, b as makeFinding, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as ExecutionProbeRequest, h as ExecutionProbeOutcome, i as AnalystFinding, l as AnalystRunResult, m as ExecutionProbe, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as ProposalFindingOrigin, w as TraceAnalysisStore, x as makeProposalFinding, y as computeFindingId } from "../types-D9ssmxKL.js";
|
|
9
|
+
import { a as createTraceAnalyst, c as BehavioralAnalystOptions, i as TraceAnalystDefinition, l as behavioralAnalyst, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as runTraceAnalyst, t as DefaultAnalystRegistryOptions, u as deriveEfficiencyFindings } from "../default-registry-G9CKMNkc.js";
|
|
10
|
+
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-qnexxJ1Z.js";
|
|
11
|
+
import { a as ExactAnalystBudgetPolicy, c as RegistryRunOpts, i as BudgetPolicy, n as AnalystRegistry, o as ExactAnalystRunExecutionError, r as AnalystRegistryOptions, s as ExactRegistryRunOpts, t as AnalystHooks } from "../registry-8You7OK1.js";
|
|
12
|
+
import { C as scoreAnalystFindings, S as traceStoreEvidenceResolver, _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, b as registryBenchmarkRunner, c as AnalystBenchmarkProvenance, d as AnalystBenchmarkSummary, f as AnalystEvidenceExpectation, g as AnalystFindingScore, h as AnalystEvidenceResolver, i as AnalystBenchmarkError, l as AnalystBenchmarkResult, m as AnalystEvidenceResolutionError, n as AnalystBenchmarkDatasetRef, o as AnalystBenchmarkObservation, p as AnalystEvidenceResolution, r as AnalystBenchmarkDescriptor, s as AnalystBenchmarkOutput, t as AnalystBenchmarkCase, u as AnalystBenchmarkRunner, v as AnalystLatencyDistribution, x as runAnalystBenchmark, y as RunAnalystBenchmarkOptions } from "../benchmark-CGPp-kDC.js";
|
|
13
|
+
import { f as ExternalOptimizerModelExecutionObservation, h as ExternalOptimizerRunnerCommand, l as ExternalOptimizerModelCall } from "../external-optimizer-contracts-szBJ_1vh.js";
|
|
14
14
|
import { t as AgentProfile } from "../agent-profile-B9_GGsG8.js";
|
|
15
|
-
import { a as
|
|
16
|
-
import { n as buildTraceToolsForGroup, t as TraceToolGroupName } from "../tool-groups-
|
|
15
|
+
import { a as resolveTraceAnalystLimits, c as RawAnalystFinding, d as parseRawFinding, i as TraceAnalystLimits, l as RawAnalystFindingSchema, n as TraceAnalysisEngineRequest, o as RAW_FINDING_SCHEMA_PROMPT, r as TraceAnalysisEngineResult, s as RawAnalystEvidence, t as TraceAnalysisEngine, u as evidenceRefsFromRawFinding } from "../engine-Cu5qD5Fc.js";
|
|
16
|
+
import { n as buildTraceToolsForGroup, t as TraceToolGroupName } from "../tool-groups-Ci8i9ErB.js";
|
|
17
17
|
import { i as nodeHttpPrimeBridgeTransport, n as PrimeBridgeTransportRequest, r as PrimeBridgeTransportResult, t as PrimeBridgeTransport } from "../prime-bridge-transport-6feEglLf.js";
|
|
18
18
|
import { z } from "zod";
|
|
19
19
|
//#region src/run-critic.d.ts
|
|
@@ -39,7 +39,6 @@ declare class RunCritic {
|
|
|
39
39
|
}
|
|
40
40
|
//#endregion
|
|
41
41
|
//#region src/analyst/adapters.d.ts
|
|
42
|
-
declare function liftSeverity(s: Severity): AnalystSeverity;
|
|
43
42
|
interface VerifierAdapterOpts<Env> {
|
|
44
43
|
id?: string;
|
|
45
44
|
area?: string;
|
|
@@ -75,11 +74,11 @@ interface SemanticConceptJudgeAdapterOpts {
|
|
|
75
74
|
id?: string;
|
|
76
75
|
area?: string;
|
|
77
76
|
/** Registry context owns cancellation and the per-analyst cost ledger. */
|
|
78
|
-
options
|
|
77
|
+
options: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>;
|
|
79
78
|
/** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */
|
|
80
79
|
settlementTimeoutMs?: number;
|
|
81
80
|
}
|
|
82
|
-
declare function createSemanticConceptJudgeAdapter(opts
|
|
81
|
+
declare function createSemanticConceptJudgeAdapter(opts: SemanticConceptJudgeAdapterOpts): Analyst<SemanticConceptJudgeInput>;
|
|
83
82
|
//#endregion
|
|
84
83
|
//#region src/analyst/benchmark-agentrx-calibration.d.ts
|
|
85
84
|
declare const AGENT_RX_UPSTREAM_REVISION = "f228165bfec60a801fd5fedd9d8ffe0f9de0c69d";
|
|
@@ -1293,25 +1292,16 @@ interface AnalystBenchmarkCommandConfig {
|
|
|
1293
1292
|
resume: boolean;
|
|
1294
1293
|
}
|
|
1295
1294
|
declare function runAnalystBenchmarkCommand(argv: readonly string[], env?: NodeJS.ProcessEnv, dependencies?: AnalystBenchmarkCommandDependencies): Promise<number>;
|
|
1296
|
-
declare const ANALYST_BENCHMARK_HELP = "agent-eval analyst-benchmark\n\nRun the recursive DSPy trace analyst against public AgentRx or CodeTraceBench labels.\n\nRequired:\n --dataset agentrx|codetracebench\n --analyst dspy-rlm|direct|prime Scored analyst. Default: dspy-rlm.\n 'direct' is the one-shot comparison arm.\n 'prime' is the RLM coding agent behind an\n OpenAI-compatible cli-bridge (codetracebench\n only; see docs/prime-analyst.md)\n --labels <dataset.json|dataset.jsonl>\n --trace-dir <one-trace-per-file OTLP JSONL directory>\n --artifact-dir <extracted artifact root> Required for CodeTraceBench\n --out <new output directory>\n --revision <full 40- or 64-character hex digest>\n --split <dataset split>\n --model-owner-module <module> dspy-rlm|direct only. Module exporting\n createModelExecutionOwner; the owner keeps\n provider credentials and policy\n --model <provider model id> For prime, the bridge model id in\n <backend>/<provider>/<model> form, e.g.\n prime/zai/glm-5.2\n --limit <positive case count>\n\nControls:\n --resume Continue an interrupted run in --out\n --bridge-url <url> prime only. OpenAI-compatible cli-bridge\n base URL. Default: http://localhost:4181\n --no-repair prime only. Disable the bounded repair turn\n for a structurally malformed reply\n --seed <integer> Case-selection and comparison seed. Default: 0\n --concurrency <positive integer> Parallel benchmark jobs. Default: 1\n --repetitions <positive integer> Runs per case and runner. Default: 1\n --rlm-samples <positive integer> Recursive-engine runs per case; above 1 the\n step-level majority consensus is scored\n (CodeTraceBench + dspy-rlm only). Default: 1\n --instructions-file <path> Replace the recursive analyst instructions\n with this file's text (dspy-rlm only). The\n recorded protocol digest binds the stock\n protocol to the override text, and\n result.json records instructionsOverrideSha256.\n --max-output-tokens <positive> Model output limit per call. Default: 16384\n --max-reasoning-tokens <integer> Reasoning-token limit per call. Default: 65536\n --max-model-requests <positive> Caller-owned model calls per analysis.\n Default: max iterations + model calls + 1\n --max-model-request-bytes <positive> Default: 16777216\n --max-model-response-bytes <positive> Default: 4194304\n --model-request-timeout-ms <positive> Default: --timeout-ms\n --max-iterations <positive> Recursive iterations per analysis. Default: 14\n --max-llm-calls <positive> DSPy model calls per analysis. Default: 8\n --max-tool-calls <positive> Trace-tool calls per analysis. Default: 80\n --max-analysis-output-chars <positive> Default: 8000\n --trace-tool-request-bytes <positive> Default: 1000000\n --trace-tool-response-bytes <positive> Default: 4000000\n --trace-tool-timeout-ms <positive> Default: 60000\n --max-process-input-bytes <positive> Default: 67108864\n --max-process-result-bytes <positive> Default: 4194304\n --max-process-output-chars <positive> Default: 64000\n --python <executable> Python with agent-eval-rpc[dspy]. Default: python\n --timeout-ms <positive> Model analyst deadline per case. Default: 300000\n --max-cost-usd <positive> Run-wide spend limit. Default: 5\n --max-artifact-bytes <positive> Final evidence bytes per case. Default: 8388608\n\nWrites result.json with every observation, metric, usage field, error, comparison,\ninput digest, artifact digest, case distribution, selected case id, and explicit\nunknown cost. Limited deterministic-hash subsets are marked non-representative.\nCompleted observations are fsynced to observations.jsonl. Shareable output is in\nresult.json and report.md. Machine-local paths, execution-owner module, and command\nare isolated in run.local.json. Provider credentials never enter this command.";
|
|
1297
1295
|
//#endregion
|
|
1298
1296
|
//#region src/analyst/benchmark-implementation.d.ts
|
|
1299
1297
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
|
|
1300
1298
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
|
|
1301
1299
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES: readonly string[];
|
|
1302
|
-
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1303
|
-
/** The published benchmark evidence was produced at this package version, by
|
|
1304
|
-
* the retired one-shot direct runner, before trace analysts moved to the
|
|
1305
|
-
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
1306
|
-
* about that artifact: the current implementation and dependency manifest have
|
|
1307
|
-
* since changed, so they cannot describe the current engine. A fresh certified
|
|
1308
|
-
* run must replace the published evidence before any accuracy number is
|
|
1309
|
-
* attributed to the engine that ships today. */
|
|
1310
|
-
declare const ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION = "0.137.0";
|
|
1300
|
+
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "6de51aaf48adb3be31601f2e53fbf06a4d407fa16a8c33156caa41bf8c21b09f";
|
|
1311
1301
|
declare const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
1312
1302
|
declare const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
1313
1303
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_FILES: readonly string[];
|
|
1314
|
-
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1304
|
+
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "6631709758e46ab562765d81bc16871fa7931ec86ba334cd85918904f3b74196";
|
|
1315
1305
|
declare function analystBenchmarkImplementationDigest(): string;
|
|
1316
1306
|
declare function analystBenchmarkDependencyLockDigest(): string;
|
|
1317
1307
|
//#endregion
|
|
@@ -1470,13 +1460,6 @@ declare function stripCodeFences(text: string): string;
|
|
|
1470
1460
|
* then `JSON.parse`. Returns `undefined` (never throws) when unrecoverable.
|
|
1471
1461
|
*/
|
|
1472
1462
|
declare function coerceJson(text: string): unknown;
|
|
1473
|
-
/**
|
|
1474
|
-
* Coerce arbitrary actor/structurer output into an array of candidate finding
|
|
1475
|
-
* rows: a JSON string → parse; a single object → 1-element array; an array →
|
|
1476
|
-
* as-is; anything else → []. Callers still run each row through Zod
|
|
1477
|
-
* (`parseRawFinding`) — this only fixes the shape and never invents fields.
|
|
1478
|
-
*/
|
|
1479
|
-
declare function coerceToFindingRows(raw: unknown): unknown[];
|
|
1480
1463
|
//#endregion
|
|
1481
1464
|
//#region src/analyst/prime-protocol.d.ts
|
|
1482
1465
|
/**
|
|
@@ -1729,5 +1712,5 @@ declare function isProposalFinding(finding: unknown): finding is ProposalFinding
|
|
|
1729
1712
|
*/
|
|
1730
1713
|
declare function assertProposalFindings(findings: unknown, context?: string): ReadonlyArray<ProposalFinding>;
|
|
1731
1714
|
//#endregion
|
|
1732
|
-
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256,
|
|
1715
|
+
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, type AgentRxBenchmarkCaseOptions, type AgentRxCalibrationRunnerSummary, type AgentRxCalibrationSummary, type AgentRxFailure, type AgentRxPrediction, type AgentRxPredictionReport, type AgentRxRow, type Analyst, type AnalystBenchmarkArtifact, type AnalystBenchmarkCase, type AnalystBenchmarkCommandConfig, type AnalystBenchmarkCommandDependencies, type AnalystBenchmarkDatasetRef, type AnalystBenchmarkDescriptor, type AnalystBenchmarkError, type AnalystBenchmarkLabelState, type AnalystBenchmarkLocalRunReceipt, type AnalystBenchmarkObservation, type AnalystBenchmarkOutput, type AnalystBenchmarkProgressRow, type AnalystBenchmarkProvenance, type AnalystBenchmarkResult, type AnalystBenchmarkRunIdentity, type AnalystBenchmarkRunManifest, type AnalystBenchmarkRunner, type AnalystBenchmarkSummary, type AnalystBudgetDeclaration, type AnalystComparisonMetric, type AnalystContext, type AnalystCost, type AnalystDefinition, type AnalystDefinitionAsymmetry, type AnalystDefinitionAsymmetryReport, type AnalystEvidenceBinding, type AnalystEvidenceExpectation, type AnalystEvidenceResolution, type AnalystEvidenceResolutionError, type AnalystEvidenceResolver, AnalystExpressivenessError, type AnalystFinding, type AnalystFindingScore, type AnalystHooks, type AnalystInputKind, type AnalystInstructionsOverride, type AnalystIssueExpectation, type AnalystLatencyDistribution, type AnalystMetricComparison, type AnalystProfileFragment, AnalystRegistry, type AnalystRegistryOptions, type AnalystRepairDeclaration, type AnalystRequirements, type AnalystRowExpansion, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystRunnerComparison, type AnalystSeverity, type AnalystTransportBinding, type AnalystUsageReceipt, type BehavioralAnalystOptions, type BudgetPolicy, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type ChunkedEvidenceBinding, type CodeTraceBenchCaseOptions, type CodeTraceBenchLabelOptions, type CodeTraceBenchLabelSet, type CodeTraceBenchRow, type CodeTraceBlockDiagnostics, type CodeTraceCalibrationRunnerSummary, type CodeTraceCalibrationSummary, type CodeTraceFailureBlock, type CodeTraceStageAnnotation, type CodeTracerLabelGroup, type CodeTracerPredictionAdapterOptions, type CodeTracerPredictions, type CodeTracerStepLabel, ControlIntegrityAnalyst, type CreateChatClientOpts, type CreateTraceAnalystOptions, type CustomTransportOpts, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, type DecodedReply, type DefaultAnalystRegistryOptions, type DefineCustomAnalystOptions, type DefineExactCustomAnalystOptions, type DiffPolicy, type DspyRlmTraceEngineOptions, type EvidenceProjection, type EvidenceRef, type ExactAnalystBudgetPolicy, type ExactAnalystBudgetSnapshot, type ExactAnalystExecutionPlanSnapshot, type ExactAnalystRunCompletion, type ExactAnalystRunEvent, ExactAnalystRunExecutionError, type ExactAnalystRunPolicySnapshot, type ExactAnalystRunResult, type ExactAnalystRunSummary, type ExactAnalystSnapshot, type ExactCapableAnalyst, type ExactExecutionComponentIdentity, type ExactExecutionComponentSnapshot, type ExactRegistryRunOpts, type ExecutionProbe, type ExecutionProbeOutcome, type ExecutionProbeRequest, type ExpandRowsArgs, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type InlineEvidenceBinding, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type LoadedVerificationArtifacts, MAX_INCORRECT_BLOCKS, MAX_INCORRECT_BLOCK_STEPS, type MockTransportOpts, type PersistedFinding, type PreparedPublicAnalystBenchmark, type PrimeBenchmarkRunnerOptions, type PrimeBridgeTransport, type PrimeBridgeTransportRequest, type PrimeBridgeTransportResult, type PrimeCodeTraceDefinitionArgs, type PrimeExchangeOptions, type PrimeExchangeOutcome, type PrimeFailure, type PrimeProjectionDelivery, type PrimeProjectionOutcome, type PrimeProjectionSource, type PrimePromptSpec, type PrimeProtocolIdentity, type PrimeRawUsage, type PrimeRejectedRow, type PrimeRepairPromptSpec, type PrimeRepairState, type PrimeReplyContract, type PrimeRowDecoded, type PrimeTurnRecord, type ProposalFinding, type ProposalFindingOrigin, type PublicAnalystBenchmarkDataset, type PublicAnalystBenchmarkModelConfig, type PublicAnalystBenchmarkModelOwner, type PublicAnalystBenchmarkModelSettings, type PublicBenchmarkDistributions, type PublicBenchmarkModelPrediction, type PublicBenchmarkSelectionReport, type PublicBenchmarkValueDistribution, type PublicDirectDefinitionArgs, type PublicRlmDefinitionArgs, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type ReplVariableConsensusPort, type ReplVariableEvidenceBinding, type ReplyContract, type ReplyEnvelope, type ReplyRowDecoded, type ReplyRowRejection, type RunAnalystBenchmarkOptions, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StepLabelAdapterOptions, type TraceAnalysisEngine, type TraceAnalysisEngineRequest, type TraceAnalysisEngineResult, type TraceAnalystDefinition, type TraceAnalystLimits, type TraceToolGroupName, type UpstreamPredictionAdapterOptions, type VerificationArtifactFile, type VerificationArtifactManifest, type VerificationArtifactRole, type VerificationAvailabilitySummary, type VerificationOutcome, type VerificationOutcomeSource, type VerificationOutcomeStatus, type VerificationResultFile, type VerifierAdapterOpts, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, analystDefinitionAsymmetries, analystDefinitionProtocolSha256, analystInstructionsOverrideFromText, analystUsageReceiptFromPrimeUsage, appendVerificationArtifactsToOtlp, assertProposalFindings, behavioralAnalyst, bindAnalyst, buildDefaultAnalystRegistry, buildPrimePrompt, buildPrimeRepairPrompt, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, compareAnalystRunners, computeFindingId, createChatClient, createDspyRlmTraceEngine, createJudgeAdapter, createPrimeBenchmarkRunner, createPublicBenchmarkDirectRunner, createPublicBenchmarkRlmRunner, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalyst, createVerifierAdapter, decodeReplyRows, defaultIsMaterial, defineCustomAnalyst, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, effectiveAnalystProtocolSha256, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPrimeRawUsage, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, expandCodeTraceFailureBlocks, extractPrimeJsonObject, findingSubjectGrammarPromptFor, isProposalFinding, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, mergePrimeRawUsage, nodeHttpPrimeBridgeTransport, normalizeAgentRxCategory, normalizeBenchmarkLabel, normalizePrimeUsage, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, primeAnalystProtocolSha256, primeCodeTraceAnalystDefinition, primeProtocolSha256, primeReplyDefect, projectPrimeTrajectory, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkRlmInstructions, publicBenchmarkSelectionReport, publicBenchmarkSystemPrompt, publicDirectAnalystDefinition, publicRlmAnalystDefinition, readAnalystBenchmarkArtifact, readAnalystInstructionsOverride, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, resolveTraceAnalystLimits, rlmEngineLimits, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, runPrimeExchange, runTraceAnalyst, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, summarizeAgentRxCalibration, summarizeAnalystBenchmarkRunner, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
1733
1716
|
//# sourceMappingURL=index.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/run-critic.ts","../../src/analyst/adapters.ts","../../src/analyst/benchmark-agentrx-calibration.ts","../../src/analyst/benchmark-dataset-types.ts","../../src/analyst/benchmark-dataset-agentrx.ts","../../src/analyst/benchmark-dataset-codetrace.ts","../../src/analyst/benchmark-dataset-utils.ts","../../src/analyst/benchmark-verification-outcome.ts","../../src/analyst/benchmark-verification-artifacts.ts","../../src/analyst/benchmark-public-types.ts","../../src/analyst/benchmark-public-adapters.ts","../../src/analyst/benchmark-public-data.ts","../../src/analyst/reply-contract.ts","../../src/analyst/definition.ts","../../src/analyst/benchmark-public-prompt.ts","../../src/analyst/benchmark-public-model.ts","../../src/analyst/benchmark-public-rlm.ts","../../src/analyst/benchmark-runner-prime.ts","../../src/analyst/benchmark-comparison.ts","../../src/analyst/benchmark-public-calibration.ts","../../src/analyst/benchmark-command-artifact.ts","../../src/analyst/benchmark-command-result.ts","../../src/analyst/benchmark-command.ts","../../src/analyst/benchmark-implementation.ts","../../src/analyst/benchmark-instructions-override.ts","../../src/analyst/benchmark-report.ts","../../src/analyst/benchmark-summary.ts","../../src/analyst/bind.ts","../../src/analyst/define.ts","../../src/analyst/kinds/skill-usage.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/prime-protocol.ts","../../src/analyst/proposal-findings.ts"],"mappings":";;;;;;;;;;;;;;;;;;;UAIiB;EACf,KAAK;EACL,OAAO;EACP,QAAQ;EACR,WAAW;EACX,QAAQ;;UAGO;EACf,UAAU,QAAQ;EAClB,gBAAgB;;cAYL;mBACM;mBACA;EAEjB,YAAY,UAAS;EAKf,MAAM,OAAO,YAAY,gBAAgB,QAAQ;EAYvD,WAAW,OAAO,WAAW;EAmH7B,KAAK,OAAO;UAIJ;;;;iBC1HM,aAAa,GAAG,WAAgB;UAe/B,oBAAoB;EACnC;EACA;EACA,UAAU,mBAAmB;;;;;EAK7B,UAAU,KAAK,cAAc;;iBAGf,sBAAsB,KAAK,MAAM,oBAAoB,OAAO,QAAQ;UAwEnE;EACf;EACA;EACA,SAAS;;EAET;;iBAGc,uBAAuB,OAAM,uBAA4B,QAAQ;UAiEhE;EACf;EACA;EACA,OAAO;;EAEP,MAAM;;EAEN,OAAO;;EAEP;;iBAGc,mBAAmB,MAAM,mBAAmB,QAAQ;UAiDnD;EACf;EACA;;EAEA,UAAU,KAAK;;EAEf;;iBAGc,kCACd,OAAM,kCACL,QAAQ;;;cC5RE;UAEI;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,SAAS;;iBAGK,4BACd,QAAQ,wBACR,2BACC;iBAkBa,iCAAiC,SAAS;;;KCtD9C;UAEK;EACf,YAAY;EACZ;EACA;EACA;EACA;EACA;;UAGe;EACf,eAAe;EACf,mBAAmB;EACnB;IAAe,YAAY;IAAY;;EACvC,wBAAwB;EACxB;EACA;EACA;;UAGe;EACf,UAAU;EACV;EACA;EACA;EACA;;UAGe;EACf,UAAU;EACV,mBAAmB;EACnB;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,iBAAiB;;KAGP,0CAEC,sCACA,iCACA;UAEI;EACf;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA,oCAAoC;;UAGrB;EACf,eAAe;EACf,WAAW,sBAAsB;;KAGvB;UAEK;;;;;EAKf,WAAW;;UAGI,kCACP,yBACN;UAEa,oCAAoC;EACnD;;EAEA;;UAGe,yCAAyC;EACxD;EACA;EACA;EACA;;UAGe,2CACP,kCACN;;;iBCtGY,qBAAqB,QACnC,KAAK,YACL,OAAO,QACP,UAAS,8BACR,qBAAqB;;iBAoGR,6BACd,mBAAmB,YACnB,iBACA,UAAS,mCACR;iBA6Ea,yBAAyB;;iBAoNzB,iBAAiB;;;iBC7YjB,mBAAmB,QACjC,KAAK,mBACL,OAAO,QACP,UAAS,4BACR,qBAAqB;;iBAwFR,gCACd,2BACA,aAAa,uBACb,UAAS,qCACR;;;iBClHa,wBAAwB;;;KCA5B;UAEK;EACf;EACA;EACA,QAAQ;;UAGO;EACf,QAAQ;EACR;EAKA;IAAe;IAAe;;EAC9B,SAAS;EACT;EACA;EACA;EACA;;UAGe;EACf;EACA;;iBA8Ec,yBACd,gBAAgB,2BACf;;;cChGU;KAED;UAEK;EACf,MAAM;EACN;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;EACT;EACA;EACA;EACA;EACA;EACA,OAAO;EACP,cAAc;EACd,UAAU,OAAO;;UAGF;EACf,UAAU;EACV,SAAS;EACT,OAAO,MAAM;IAA6B;;;iBAatB,mCAAmC;EACvD;EACA,KAAK;EACL;IACE,QAAQ;iBAwHI,kCACd,kBACA,iBACA,WAAW,6BACX;;;KC5KU;;;;;;;;UASK;;WAEN;;WAEA;;;UAIM;EACf,MAAM;EACN;EACA,kBAAkB,aAAa;;EAE/B,UAAU;;UAGK;;EAEf,MAAM;;EAEN;;EAEA,kBAAkB,aAAa;EAC/B;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,UAAU;;EAEV;;EAEA,uBAAuB;EACvB;IACE,SAAS;IACT;IACA;IACA;IACA;IACA;IACA;IACA;IACA;;;;;;IAMA;;EAEF,aAAa;EACb;IACE;IACA;;;;;;;;;KAUQ,sCAAsC,KAChD,iEAGA,QAAQ,KAAK;UAEE;EACf,OAAO,qBAAqB;EAC5B;EACA;EACA;EACA,YAAY;IACV;IACA;IACA;;EAEF,uBAAuB;EACvB,WAAW;;UAGI;EACf;EACA;EACA,QAAQ;;UAGO;EACf,OAAO;EACP,OAAO;EACP,OAAO;EACP,YAAY;EACZ,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;EACA;EACA,QAAQ;EACR,UAAU;;;;;;;;;;UChHK;EACf;EACA;;EAEA;EACA;EACA,UAAU;EACV;EACA;EACA;EACA;EACA,WAAW;;;;;;;;;UAUI;EACf;EACA;;EAEA,kCAAkC;;EAElC;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;UASe;EACf;EACA,OAAO;;iBAGO,8BAA8B,uBAAuB;iBAiB/C,6BAA6B;EACjD,SAAS;EACT;EACA,mBAAmB;EACnB;EACA,OAAO;EACP,SAAS;IACP;EACF,UAAU;EACV,aAAa;;EAEb,aAAa;;;;;;;;;iBA4MO,6BAA6B;EACjD;EACA,iBAAiB;EACjB,OAAO;EACP;EACA;EACA,SAAS;IACP;EACF,UAAU;EACV,aAAa;EACb,YAAY;;;;iBCtQQ,wBACpB,eACC,QAAQ,MAAM;iBA2BD,0BACd,SAAS,+BACT,eAAe,2BACf;EAAW;EAAe;IACzB,MAAM;iBAwBO,6BACd,SAAS,+BACT,eAAe,4BACd;iBAyCa,+BACd,SAAS,+BACT,iBAAiB,2BACjB,mBAAmB,2BACnB,eACC;iBAcmB,8BAA8B;EAClD,SAAS;EACT;EACA;EACA;EACA;EACA;EACA;IACE,QAAQ;;;;;;;;;;;;;KCnKA,gBAAgB;EAAU;EAAU,KAAK;;EAAW;EAAW;;;UAG1D;EACf;;EAEA,QAAQ;;UAGO,cAAc;;EAE7B;;EAEA;;;;;EAKA;;;;;;EAMA,UAAU,cAAc,gBAAgB,gBAAgB;;;;;;EAMxD;;;;;;EAMA,eAAe,iBAAiB;;;;;;EAMhC;;;;;EAKA;;;UAIe;EACf;EACA;;UAGe,aAAa;EAC5B,MAAM;;EAEN,QAAQ;EACR,UAAU;;EAEV;;EAEA;;;;;;iBAmCc,gBAAgB,MAC9B,UAAU,cAAc,OACxB,iBACC,aAAa;;;;;;;;KCrEJ,yBAAyB,KAAK;;;;;;KAS9B;WAEG;;WAEA;;;;;;WAMA;;WAGA;;WAEA;;;WAIA;WACA,WAAW;;;WAIX;WACA,WAAW;;UAKT;;WAEN;;WAEA;;WAEA;;WAEA,eAAe;;UAGT;;;;;WAKN;;UAKM;EACf,UAAU;;EAEV;;UAGe,eAAe;;EAE9B;EACA,eAAe;EACf,OAAO;EACP;EACA;;EAEA;EACA,SAAS;;;UAIM,sBAAsB;WAC5B;EACT,kBAAkB;;WAET,cAAc,SAAS;;;;;EAKhC,OAAO,iBAAiB,gBAAgB;;EAExC,QAAQ,iBAAiB,gBAAgB;EACzC,WAAW,MAAM,eAAe,QAAQ,QAAQ;;;UAIjC,uBAAuB;WAC7B;EACT,kBAAkB;WACT,cAAc,SAAS;;WAEvB;;WAEA;;EAET,YAAY;EACZ,WAAW,MAAM,eAAe,QAAQ,QAAQ;;EAEhD,gBAAgB;IACd;IACA,mBAAmB;IACnB,OAAO;IACP,SAAS;MACP;;;UAIW,0BAA0B,aAAa;;EAEtD,KAAK,SAAS,uBAAuB;IACnC,iBAAiB;IACjB;;;EAGF,OAAO;IACL;IACA,iBAAiB;IACjB,OAAO;IACP;IACA;IACA,SAAS;MACP,QAAQ;;EAEZ,aAAa,sBAAsB,gBAAgB;;;UAIpC,4BAA4B,uBAAuB;WACzD;;WAEA;EACT,kBAAkB;WACT,cAAc,SAAS;;WAEvB,qBAAqB,SAAS;;WAE9B;;EAET,qBAAqB,8BAA8B;;EAEnD,MAAM;IACJ;IACA,mBAAmB;IACnB;IACA,OAAO;IACP,SAAS;MACP;IAAU,UAAU;IAAkB,aAAa;IAAe;;;EAEtE,YAAY,0BAA0B,aAAa;;EAEnD,mBACE,QAAQ,oCACP,uBAAuB;;KAGhB,uBAAuB,MAAM,uBAAuB,oBAC5D,sBAAsB,QACtB,uBAAuB,QACvB,4BAA4B,aAAa;UAI5B,kBAAkB,gBAAgB,uBAAuB;;WAE/D;WACA;WACA;;WAEA;WACA,SAAS;;WAET;;WAEA;WACA,YAAY;WACZ,eAAe,cAAc;;;;;;WAM7B,gBAAgB,SAAS;WACzB,QAAQ;WACR,QAAQ;;;;;;WAMR;WACA,SAAS,uBAAuB,MAAM,aAAa;;;;;;;;;cAUjD,mCAAmC;;;;;;;iBAUhC,gCAAgC,MAAM,aAAa,QACjE,YAAY,kBAAkB,MAAM,aAAa;;UAwClC;WACN;WACA,gBAAgB;WAChB,iBAAiB,YAAY;WAC7B;WACA;WACA;;WAEA;;WAEA;;UAGM;WACN;;WAEA;;WAEA,sBAAsB;WACtB,sBAAsB;;;;;;;;;iBAUjB,6BACd,aAAa,cAAc,gDAC1B;;;;;;;cCzUU;;;;cAKA;cAMA;;iBA8GG,4BAA4B,SAAS;;iBASrC,+BAA+B,SAAS;;;;iBAcxC,8BAA8B,SAAS;;;;;;;;;;;;;UC5EtC;;EAEf;;EAEA;;EAEA;;;iBAIc,8BACd,SAAS,+BACT,MAAM,6BACL,kBAAkB;;iBAsEL,kCACd,SAAS,+BACT,QAAQ,oCACP,uBAAuB;cAmhBpB,yBAAuB,EAAA;;;;;;;;;;;;;GASlB,EAAA,KAAA;cACL,gCAA8B,EAAA;;;;;;;;;;;;;;;;;;;GAsChC,EAAA,KAAA;cACE,uBAAqB,EAAA;;;;;;;;;;;;KA2BtB,yBAAyB,EAAE,aAAa;EAC3C,WAAW,EAAE,aAAa;;KAEvB,2BAA2B,EAAE,aAAa;KACnC,iCAAiC,yBAAyB;;;;;;;;;;;;;UC5sBrD;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,cAAc;;;iBAIA,2BACd,SAAS,+BACT,MAAM,0BACL,kBAAkB,mBAAmB;;iBA+GxB,+BACd,SAAS,+BACT,QAAQ,oCACP,uBAAuB;;iBAwBV,gBAAgB,QAAQ,oCAAoC;;;UCtI3D;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA,UAAU;;EAEV,YAAY;;;;;;iBA8EE;UAIC;;EAEf;;EAEA;;;;;;iBAOc,gCACd,MAAM,+BACL,kBAAkB;;iBAyDL,2BACd,SAAS,8BACR,uBAAuB;;;KCnPd;UAoBK;EACf,QAAQ;EACR;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;;iBASK,sBACd,QAAQ,wBACR;EACE;EACA;EACA;EACA;EACA;IAED;;;UCnEc;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;;iBAGK,8BACd,QAAQ,yBACP;iBAca,mCAAmC,SAAS;;;UCpC3C;EACf;EACA;EACA;IACE,SAAS;IACT;IACA;IACA;IACA;IACA,YAAY;MAAQ;MAAiB;MAAsB;;IAC3D,uBAAuB;IACvB,0BAA0B;IAC1B;MACE;MACA;MACA;MACA,QAAQ;;IAEV;MACE;MACA;;MAEA;MACA;;MAEA;MACA;MACA;MACA;MACA;MACA;MACA;MACA,UAAU;MACV;QACE;QACA;QACA;QACA;QACA;QACA;QACA;QACA;;MAEF;QACE;QACA;QACA;;MAEF;MACA;MACA;;MAEA;MACA;MACA;;;EAGJ,QAAQ;EACR,aAAa;EACb,uBAAuB;EACvB,qBAAqB;;UAGN;EACf;EACA;EACA;EACA;IACE;IACA;IACA;;;UAIa;EACf;IACE,SAAS;IACT;IACA;IACA;MACE;MACA;MACA;MACA;MACA;MACA;MACA;MACA;MACA,SAAS;MACT;QACE;QACA;QACA;QACA;QACA;QACA;QACA;QACA;;MAEF;QACE;QACA;QACA;;;IAGJ;IACA;IACA;IACA;;IAEA;IACA;IACA;IACA;;IAEA;IACA;IACA;IACA;;EAEF;IACE;IACA;IACA;IACA,YAAY;MAAQ;MAAiB;MAAsB;;IAC3D;IACA;;;UAIa;EACf;EACA;EACA;EACA;EACA,UAAU;;UAGK;EACf;EACA;EACA;EACA;IACE;IACA;IACA;IACA;;IAEA;;EAEF;EACA;IACE;IACA;IACA;;EAEF;IACE;IACA;IACA;IACA;IACA;IACA;;;UAIa;EACf;EACA;EACA;EACA,aAAa;EACb;;cAGW;cACA;cACA;cACA;;;iBC5KS,6BACpB,eACC,QAAQ;;;UCuEM;EACf,uBACE,SAAS,+BACT,QAAQ,wCACL,uBAAuB;EAC5B,2BACE,mBACA;IACE;IACA,aAAa,SAAS,OAAO;QAE5B,QAAQ;;;;;;;;;KAUH;;UAGK;EACf;EACA;;UAGe;EACf,SAAS;EACT,SAAS;EACT;EACA;EACA;EACA;EACA;EACA;EACA,OAAO;EACP;EACA;EACA;EACA;;EAEA;EACA;EACA;;EAEA;;EAEA,QAAQ;EACR;EACA;;iBAKoB,2BACpB,yBACA,MAAK,OAAO,YACZ,eAAc,sCACb;cAsUU;;;cC/dA;cAEA;cAEA;cAOA;;;;;;;;cAUA;cAEA;cAGA;cAGA;cA6GA;iBAGG;iBAIA;;;;iBCxIA,oCAAoC,eAAe;;iBAQnD,gCAAgC,eAAe;;;;;;;;;;;iBAyB/C,+BACd,SAAS,+BACT,WAAW,KAAK;;;iBCzCF,+BACd,QAAQ,wBACR,uBAAsB;;;iBCER,gCACd,kBACA,uBAAuB,gCACtB;;;;KCgBS;;WAGG;WACA;WACA;WACA,YAAY;WACZ,UAAU;;;WAIV;WACA,QAAQ;;;;;;iBAOP,YAAY,MAAM,aAAa,QAC7C,YAAY,kBAAkB,MAAM,aAAa,SACjD,YAAY,0BACX,uBAAuB;;;;;;;;;;iBCpCV,mBAAmB,YAAY,yBAAyB;UAkBvD;EACf;EACA;EACA;EACA,MAAM;EACN,SAAS,QAAQ;;UAGF,wCAAwC;;EAEvD,iBAAiB,SAAS;;;iBAIZ,oBACd,SAAS,kCACR,oBAAoB;iBACP,oBACd,SAAS,6BACR,QAAQ;;;KCpBC;;UAGK;EACf;EACA,MAAM;;EAEN;EACA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;EACA;EACA;;EAEA;;EAEA;;UAGe;EACf;EACA,SAAS;;UAGM;;EAEf;;EAEA;IAAc;IAAc,MAAM;;;EAElC;;;EAGA,kBAAkB;;EAElB;;;iBA2Ec,sBAAsB,QAAQ,uBAAuB;;iBAiHrD,uBACd,QAAQ,kBACR,qBACC;cA2HU,6BAA6B,oBAAoB;WACnD;WACA;WAEA;WACA;IAAS;IAAgC;;WACzC;WACA;aACP;aACA;aACA;;EAGI,QAAQ,OAAO,kBAAkB,KAAK,iBAAiB,QAAQ;;cAS1D,qBAAmB;;;;;;;;;;;;;;;iBC5YhB,gBAAgB;;;;;iBAgBhB,WAAW;;;;;;;iBAeX,oBAAoB;;;;;;;;;;;;;;;;;;;;;;;;;;;;KCVxB,gBAAgB,QAAQ,gBAAgB;KAExC,mBAAmB,QAAQ,cAAc;UAIpC;EACf;;EAEA;EACA;;EAEA;EACA;;EAEA;;iBAGc,iBAAiB,MAAM;UAatB;EACf;EACA;EACA;;;iBAIc,uBAAuB,MAAM;;;;;;;;;;iBAuB7B,uBAAuB,eAAe;;iBAkBtC,iBACd,QAAQ,gCACR;;;;;;UAyBe;;;;;;EAMf;;EAEA;;EAEA;;;;;;;EAOA;;iBAGc,sBAAsB;;iBAKtB,oBAAoB,eAAe;;;;;;iBAgBnC,mBAAmB,GAAG,eAAe,GAAG,gBAAgB;;;;;;;;;;;;;iBA6BxD,kCACd,OAAO,eACP,SAAS,qBACR;;;;;;;;KAqCS;EAEN;EACA;;EAEA;EAAqB;EAAiB;EAAgB;;EACtD;EAAiB;EAAiB;;UAEvB;EACf;;EAEA;;;UAIe;EACf;EACA,OAAO;;EAEP;;UAGe;EACf;EACA;;KAGU,qBAAqB;EAE3B;;EAEA;EACA,MAAM;EACN,UAAU;;EAEV;;EAEA;EACA,OAAO;EACP,OAAO;EACP,QAAQ;EACR;;EAGA;EACA,SAAS;EACT,OAAO;EACP,OAAO;EACP,QAAQ;;EAER;;UAGW,qBAAqB;EACpC,UAAU,mBAAmB;EAC7B;EACA,WAAW;;EAEX;EACA;;EAEA;;EAEA;EACA,SAAS;;;;;;;iBAQW,iBAAiB,MACrC,SAAS,qBAAqB,QAC7B,QAAQ,qBAAqB;;;;;;UAyLf,sBAAsB;;EAErC,QAAQ,iBAAiB;;EAEzB,UAAU,iBAAiB;;EAE3B;;UAGe;EACf;EACA;EACA;;KAGU,uBAAuB;EAE7B;EACA,gBAAgB;EAChB;EACA,UAAU;;EAEV;EAAW;EAAgB;;;;;;;;;iBASX,uBAAuB,OAC3C,QAAQ,sBAAsB,QAC9B;EAAU;IACT,QAAQ,uBAAuB;UA8BjB;EACf;EACA;EACA;EACA;;EAEA,QAAQ,SAAS;;;;;;;;;;;;iBAaH,oBAAoB,UAAU;;;;iBC/iB9B,kBAAkB,mBAAmB,WAAW;;;;;;iBAShD,uBACd,mBACA,mBACC,cAAc"}
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/run-critic.ts","../../src/analyst/adapters.ts","../../src/analyst/benchmark-agentrx-calibration.ts","../../src/analyst/benchmark-dataset-types.ts","../../src/analyst/benchmark-dataset-agentrx.ts","../../src/analyst/benchmark-dataset-codetrace.ts","../../src/analyst/benchmark-dataset-utils.ts","../../src/analyst/benchmark-verification-outcome.ts","../../src/analyst/benchmark-verification-artifacts.ts","../../src/analyst/benchmark-public-types.ts","../../src/analyst/benchmark-public-adapters.ts","../../src/analyst/benchmark-public-data.ts","../../src/analyst/reply-contract.ts","../../src/analyst/definition.ts","../../src/analyst/benchmark-public-prompt.ts","../../src/analyst/benchmark-public-model.ts","../../src/analyst/benchmark-public-rlm.ts","../../src/analyst/benchmark-runner-prime.ts","../../src/analyst/benchmark-comparison.ts","../../src/analyst/benchmark-public-calibration.ts","../../src/analyst/benchmark-command-artifact.ts","../../src/analyst/benchmark-command-result.ts","../../src/analyst/benchmark-command.ts","../../src/analyst/benchmark-implementation.ts","../../src/analyst/benchmark-instructions-override.ts","../../src/analyst/benchmark-report.ts","../../src/analyst/benchmark-summary.ts","../../src/analyst/bind.ts","../../src/analyst/define.ts","../../src/analyst/kinds/skill-usage.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/prime-protocol.ts","../../src/analyst/proposal-findings.ts"],"mappings":";;;;;;;;;;;;;;;;;;;UAIiB;EACf,KAAK;EACL,OAAO;EACP,QAAQ;EACR,WAAW;EACX,QAAQ;;UAGO;EACf,UAAU,QAAQ;EAClB,gBAAgB;;cAYL;mBACM;mBACA;EAEjB,YAAY,UAAS;EAKf,MAAM,OAAO,YAAY,gBAAgB,QAAQ;EAYvD,WAAW,OAAO,WAAW;EAmH7B,KAAK,OAAO;UAIJ;;;;UC3GO,oBAAoB;EACnC;EACA;EACA,UAAU,mBAAmB;;;;;EAK7B,UAAU,KAAK,cAAc;;iBAGf,sBAAsB,KAAK,MAAM,oBAAoB,OAAO,QAAQ;UAwEnE;EACf;EACA;EACA,SAAS;;EAET;;iBAGc,uBAAuB,OAAM,uBAA4B,QAAQ;UAiEhE;EACf;EACA;EACA,OAAO;;EAEP,MAAM;;EAEN,OAAO;;EAEP;;iBAGc,mBAAmB,MAAM,mBAAmB,QAAQ;UAiDnD;EACf;EACA;;EAEA,SAAS,KAAK;;EAEd;;iBAGc,kCACd,MAAM,kCACL,QAAQ;;;cC5RE;UAEI;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,SAAS;;iBAGK,4BACd,QAAQ,wBACR,2BACC;iBAkBa,iCAAiC,SAAS;;;KCtD9C;UAEK;EACf,YAAY;EACZ;EACA;EACA;EACA;EACA;;UAGe;EACf,eAAe;EACf,mBAAmB;EACnB;IAAe,YAAY;IAAY;;EACvC,wBAAwB;EACxB;EACA;EACA;;UAGe;EACf,UAAU;EACV;EACA;EACA;EACA;;UAGe;EACf,UAAU;EACV,mBAAmB;EACnB;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,iBAAiB;;KAGP,0CAEC,sCACA,iCACA;UAEI;EACf;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA,oCAAoC;;UAGrB;EACf,eAAe;EACf,WAAW,sBAAsB;;KAGvB;UAEK;;;;;EAKf,WAAW;;UAGI,kCACP,yBACN;UAEa,oCAAoC;EACnD;;EAEA;;UAGe,yCAAyC;EACxD;EACA;EACA;EACA;;UAGe,2CACP,kCACN;;;iBCtGY,qBAAqB,QACnC,KAAK,YACL,OAAO,QACP,UAAS,8BACR,qBAAqB;;iBAsGR,6BACd,mBAAmB,YACnB,iBACA,UAAS,mCACR;iBA6Ea,yBAAyB;;iBAoNzB,iBAAiB;;;iBC/YjB,mBAAmB,QACjC,KAAK,mBACL,OAAO,QACP,UAAS,4BACR,qBAAqB;;iBAwFR,gCACd,2BACA,aAAa,uBACb,UAAS,qCACR;;;iBClHa,wBAAwB;;;KCA5B;UAEK;EACf;EACA;EACA,QAAQ;;UAGO;EACf,QAAQ;EACR;EAKA;IAAe;IAAe;;EAC9B,SAAS;EACT;EACA;EACA;EACA;;UAGe;EACf;EACA;;iBA8Ec,yBACd,gBAAgB,2BACf;;;cChGU;KAED;UAEK;EACf,MAAM;EACN;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;EACT;EACA;EACA;EACA;EACA;EACA,OAAO;EACP,cAAc;EACd,UAAU,OAAO;;UAGF;EACf,UAAU;EACV,SAAS;EACT,OAAO,MAAM;IAA6B;;;iBAatB,mCAAmC;EACvD;EACA,KAAK;EACL;IACE,QAAQ;iBAwHI,kCACd,kBACA,iBACA,WAAW,6BACX;;;KC5KU;;;;;;;;UASK;;WAEN;;WAEA;;;UAIM;EACf,MAAM;EACN;EACA,kBAAkB,aAAa;;EAE/B,UAAU;;UAGK;;EAEf,MAAM;;EAEN;;EAEA,kBAAkB,aAAa;EAC/B;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,UAAU;;EAEV;;EAEA,uBAAuB;EACvB;IACE,SAAS;IACT;IACA;IACA;IACA;IACA;IACA;IACA;IACA;;;;;;IAMA;;EAEF,aAAa;EACb;IACE;IACA;;;;;;;;;KAUQ,sCAAsC,KAChD,iEAGA,QAAQ,KAAK;UAEE;EACf,OAAO,qBAAqB;EAC5B;EACA;EACA;EACA,YAAY;IACV;IACA;IACA;;EAEF,uBAAuB;EACvB,WAAW;;UAGI;EACf;EACA;EACA,QAAQ;;UAGO;EACf,OAAO;EACP,OAAO;EACP,OAAO;EACP,YAAY;EACZ,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;EACA;EACA,QAAQ;EACR,UAAU;;;;;;;;;;UChHK;EACf;EACA;;EAEA;EACA;EACA,UAAU;EACV;EACA;EACA;EACA;EACA,WAAW;;;;;;;;;UAUI;EACf;EACA;;EAEA,kCAAkC;;EAElC;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;UASe;EACf;EACA,OAAO;;iBAGO,8BAA8B,uBAAuB;iBAiB/C,6BAA6B;EACjD,SAAS;EACT;EACA,mBAAmB;EACnB;EACA,OAAO;EACP,SAAS;IACP;EACF,UAAU;EACV,aAAa;;EAEb,aAAa;;;;;;;;;iBA4MO,6BAA6B;EACjD;EACA,iBAAiB;EACjB,OAAO;EACP;EACA;EACA,SAAS;IACP;EACF,UAAU;EACV,aAAa;EACb,YAAY;;;;iBCtQQ,wBACpB,eACC,QAAQ,MAAM;iBA2BD,0BACd,SAAS,+BACT,eAAe,2BACf;EAAW;EAAe;IACzB,MAAM;iBAwBO,6BACd,SAAS,+BACT,eAAe,4BACd;iBAyCa,+BACd,SAAS,+BACT,iBAAiB,2BACjB,mBAAmB,2BACnB,eACC;iBAcmB,8BAA8B;EAClD,SAAS;EACT;EACA;EACA;EACA;EACA;EACA;IACE,QAAQ;;;;;;;;;;;;;KCnKA,gBAAgB;EAAU;EAAU,KAAK;;EAAW;EAAW;;;UAG1D;EACf;;EAEA,QAAQ;;UAGO,cAAc;;EAE7B;;EAEA;;;;;EAKA;;;;;;EAMA,UAAU,cAAc,gBAAgB,gBAAgB;;;;;;EAMxD;;;;;;EAMA,eAAe,iBAAiB;;;;;;EAMhC;;;;;EAKA;;;UAIe;EACf;EACA;;UAGe,aAAa;EAC5B,MAAM;;EAEN,QAAQ;EACR,UAAU;;EAEV;;EAEA;;;;;;iBAmCc,gBAAgB,MAC9B,UAAU,cAAc,OACxB,iBACC,aAAa;;;;;;;;KCrEJ,yBAAyB,KAAK;;;;;;KAS9B;WAEG;;WAEA;;;;;;WAMA;;WAGA;;WAEA;;;WAIA;WACA,WAAW;;;WAIX;WACA,WAAW;;UAKT;;WAEN;;WAEA;;WAEA;;WAEA,eAAe;;UAGT;;;;;WAKN;;UAKM;EACf,UAAU;;EAEV;;UAGe,eAAe;;EAE9B;EACA,eAAe;EACf,OAAO;EACP;EACA;;EAEA;EACA,SAAS;;;UAIM,sBAAsB;WAC5B;EACT,kBAAkB;;WAET,cAAc,SAAS;;;;;EAKhC,OAAO,iBAAiB,gBAAgB;;EAExC,QAAQ,iBAAiB,gBAAgB;EACzC,WAAW,MAAM,eAAe,QAAQ,QAAQ;;;UAIjC,uBAAuB;WAC7B;EACT,kBAAkB;WACT,cAAc,SAAS;;WAEvB;;WAEA;;EAET,YAAY;EACZ,WAAW,MAAM,eAAe,QAAQ,QAAQ;;EAEhD,gBAAgB;IACd;IACA,mBAAmB;IACnB,OAAO;IACP,SAAS;MACP;;;UAIW,0BAA0B,aAAa;;EAEtD,KAAK,SAAS,uBAAuB;IACnC,iBAAiB;IACjB;;;EAGF,OAAO;IACL;IACA,iBAAiB;IACjB,OAAO;IACP;IACA;IACA,SAAS;MACP,QAAQ;;EAEZ,aAAa,sBAAsB,gBAAgB;;;UAIpC,4BAA4B,uBAAuB;WACzD;;WAEA;EACT,kBAAkB;WACT,cAAc,SAAS;;WAEvB,qBAAqB,SAAS;;WAE9B;;EAET,qBAAqB,8BAA8B;;EAEnD,MAAM;IACJ;IACA,mBAAmB;IACnB;IACA,OAAO;IACP,SAAS;MACP;IAAU,UAAU;IAAkB,aAAa;IAAe;;;EAEtE,YAAY,0BAA0B,aAAa;;EAEnD,mBACE,QAAQ,oCACP,uBAAuB;;KAGhB,uBAAuB,MAAM,uBAAuB,oBAC5D,sBAAsB,QACtB,uBAAuB,QACvB,4BAA4B,aAAa;UAI5B,kBAAkB,gBAAgB,uBAAuB;;WAE/D;WACA;WACA;;WAEA;WACA,SAAS;;WAET;;WAEA;WACA,YAAY;WACZ,eAAe,cAAc;;;;;;WAM7B,gBAAgB,SAAS;WACzB,QAAQ;WACR,QAAQ;;;;;;WAMR;WACA,SAAS,uBAAuB,MAAM,aAAa;;;;;;;;;cAUjD,mCAAmC;;;;;;;iBAUhC,gCAAgC,MAAM,aAAa,QACjE,YAAY,kBAAkB,MAAM,aAAa;;UAwClC;WACN;WACA,gBAAgB;WAChB,iBAAiB,YAAY;WAC7B;WACA;WACA;;WAEA;;WAEA;;UAGM;WACN;;WAEA;;WAEA,sBAAsB;WACtB,sBAAsB;;;;;;;;;iBAUjB,6BACd,aAAa,cAAc,gDAC1B;;;;;;;cCzUU;;;;cAKA;cAMA;;iBA8GG,4BAA4B,SAAS;;iBASrC,+BAA+B,SAAS;;;;iBAcxC,8BAA8B,SAAS;;;;;;;;;;;;;UC5EtC;;EAEf;;EAEA;;EAEA;;;iBAIc,8BACd,SAAS,+BACT,MAAM,6BACL,kBAAkB;;iBAsEL,kCACd,SAAS,+BACT,QAAQ,oCACP,uBAAuB;cAwhBpB,yBAAuB,EAAA;;;;;;;;;;;;;GASlB,EAAA,KAAA;cACL,gCAA8B,EAAA;;;;;;;;;;;;;;;;;;;GAsChC,EAAA,KAAA;cACE,uBAAqB,EAAA;;;;;;;;;;;;KA2BtB,yBAAyB,EAAE,aAAa;EAC3C,WAAW,EAAE,aAAa;;KAEvB,2BAA2B,EAAE,aAAa;KACnC,iCAAiC,yBAAyB;;;;;;;;;;;;;UCjtBrD;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,cAAc;;;iBAIA,2BACd,SAAS,+BACT,MAAM,0BACL,kBAAkB,mBAAmB;;iBA+GxB,+BACd,SAAS,+BACT,QAAQ,oCACP,uBAAuB;;iBAwBV,gBAAgB,QAAQ,oCAAoC;;;UCtI3D;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA,UAAU;;EAEV,YAAY;;;;;;iBA8EE;UAIC;;EAEf;;EAEA;;;;;;iBAOc,gCACd,MAAM,+BACL,kBAAkB;;iBAyDL,2BACd,SAAS,8BACR,uBAAuB;;;KCnPd;UAoBK;EACf,QAAQ;EACR;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;;iBASK,sBACd,QAAQ,wBACR;EACE;EACA;EACA;EACA;EACA;IAED;;;UCnEc;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;;iBAGK,8BACd,QAAQ,yBACP;iBAca,mCAAmC,SAAS;;;UCpC3C;EACf;EACA;EACA;IACE,SAAS;IACT;IACA;IACA;IACA;IACA,YAAY;MAAQ;MAAiB;MAAsB;;IAC3D,uBAAuB;IACvB,0BAA0B;IAC1B;MACE;MACA;MACA;MACA,QAAQ;;IAEV;MACE;MACA;;MAEA;MACA;;MAEA;MACA;MACA;MACA;MACA;MACA;MACA;MACA,UAAU;MACV;QACE;QACA;QACA;QACA;QACA;QACA;QACA;QACA;;MAEF;QACE;QACA;QACA;;MAEF;MACA;MACA;;MAEA;MACA;MACA;;;EAGJ,QAAQ;EACR,aAAa;EACb,uBAAuB;EACvB,qBAAqB;;UAGN;EACf;EACA;EACA;EACA;IACE;IACA;IACA;;;UAIa;EACf;IACE,SAAS;IACT;IACA;IACA;MACE;MACA;MACA;MACA;MACA;MACA;MACA;MACA;MACA,SAAS;MACT;QACE;QACA;QACA;QACA;QACA;QACA;QACA;QACA;;MAEF;QACE;QACA;QACA;;;IAGJ;IACA;IACA;IACA;;IAEA;IACA;IACA;IACA;;IAEA;IACA;IACA;IACA;;EAEF;IACE;IACA;IACA;IACA,YAAY;MAAQ;MAAiB;MAAsB;;IAC3D;IACA;;;UAIa;EACf;EACA;EACA;EACA;EACA,UAAU;;UAGK;EACf;EACA;EACA;EACA;IACE;IACA;IACA;IACA;;IAEA;;EAEF;EACA;IACE;IACA;IACA;;EAEF;IACE;IACA;IACA;IACA;IACA;IACA;;;UAIa;EACf;EACA;EACA;EACA,aAAa;EACb;;cAGW;cACA;cACA;cACA;;;iBC5KS,6BACpB,eACC,QAAQ;;;UCuEM;EACf,uBACE,SAAS,+BACT,QAAQ,wCACL,uBAAuB;EAC5B,2BACE,mBACA;IACE;IACA,aAAa,SAAS,OAAO;QAE5B,QAAQ;;;;;;;;;KAUH;;UAGK;EACf;EACA;;UAGe;EACf,SAAS;EACT,SAAS;EACT;EACA;EACA;EACA;EACA;EACA;EACA,OAAO;EACP;EACA;EACA;EACA;;EAEA;EACA;EACA;;EAEA;;EAEA,QAAQ;EACR;EACA;;iBAKoB,2BACpB,yBACA,MAAK,OAAO,YACZ,eAAc,sCACb;;;cCzJU;cAEA;cAEA;cAOA;cAYA;cAGA;cAGA;cA8GA;iBAGG;iBAIA;;;;iBCzIA,oCAAoC,eAAe;;iBAQnD,gCAAgC,eAAe;;;;;;;;;;;iBAyB/C,+BACd,SAAS,+BACT,WAAW,KAAK;;;iBCzCF,+BACd,QAAQ,wBACR,uBAAsB;;;iBCER,gCACd,kBACA,uBAAuB,gCACtB;;;;KCgBS;;WAGG;WACA;WACA;WACA,YAAY;WACZ,UAAU;;;WAIV;WACA,QAAQ;;;;;;iBAOP,YAAY,MAAM,aAAa,QAC7C,YAAY,kBAAkB,MAAM,aAAa,SACjD,YAAY,0BACX,uBAAuB;;;;;;;;;;iBCpCV,mBAAmB,YAAY,yBAAyB;UAkBvD;EACf;EACA;EACA;EACA,MAAM;EACN,SAAS,QAAQ;;UAGF,wCAAwC;;EAEvD,iBAAiB,SAAS;;;iBAIZ,oBACd,SAAS,kCACR,oBAAoB;iBACP,oBACd,SAAS,6BACR,QAAQ;;;KCpBC;;UAGK;EACf;EACA,MAAM;;EAEN;EACA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;EACA;EACA;;EAEA;;EAEA;;UAGe;EACf;EACA,SAAS;;UAGM;;EAEf;;EAEA;IAAc;IAAc,MAAM;;;EAElC;;;EAGA,kBAAkB;;EAElB;;;iBA2Ec,sBAAsB,QAAQ,uBAAuB;;iBAiHrD,uBACd,QAAQ,kBACR,qBACC;cA2HG,6BAA6B,oBAAoB;WAC5C;WACA;WAEA;WACA;IAAS;IAAgC;;WACzC;WACA;aACP;aACA;aACA;;EAGI,QAAQ,OAAO,kBAAkB,KAAK,iBAAiB,QAAQ;;cAS1D,qBAAmB;;;;;;;;;;;;;;;iBC5YhB,gBAAgB;;;;;iBAgBhB,WAAW;;;;;;;;;;;;;;;;;;;;;;;;;;;;KCKf,gBAAgB,QAAQ,gBAAgB;KAExC,mBAAmB,QAAQ,cAAc;UAIpC;EACf;;EAEA;EACA;;EAEA;EACA;;EAEA;;iBAGc,iBAAiB,MAAM;UAatB;EACf;EACA;EACA;;;iBAIc,uBAAuB,MAAM;;;;;;;;;;iBAuB7B,uBAAuB,eAAe;;iBAkBtC,iBACd,QAAQ,gCACR;;;;;;UAyBe;;;;;;EAMf;;EAEA;;EAEA;;;;;;;EAOA;;iBAGc,sBAAsB;;iBAKtB,oBAAoB,eAAe;;;;;;iBAgBnC,mBAAmB,GAAG,eAAe,GAAG,gBAAgB;;;;;;;;;;;;;iBA6BxD,kCACd,OAAO,eACP,SAAS,qBACR;;;;;;;;KAqCS;EAEN;EACA;;EAEA;EAAqB;EAAiB;EAAgB;;EACtD;EAAiB;EAAiB;;UAEvB;EACf;;EAEA;;;UAIe;EACf;EACA,OAAO;;EAEP;;UAGe;EACf;EACA;;KAGU,qBAAqB;EAE3B;;EAEA;EACA,MAAM;EACN,UAAU;;EAEV;;EAEA;EACA,OAAO;EACP,OAAO;EACP,QAAQ;EACR;;EAGA;EACA,SAAS;EACT,OAAO;EACP,OAAO;EACP,QAAQ;;EAER;;UAGW,qBAAqB;EACpC,UAAU,mBAAmB;EAC7B;EACA,WAAW;;EAEX;EACA;;EAEA;;EAEA;EACA,SAAS;;;;;;;iBAQW,iBAAiB,MACrC,SAAS,qBAAqB,QAC7B,QAAQ,qBAAqB;;;;;;UAyLf,sBAAsB;;EAErC,QAAQ,iBAAiB;;EAEzB,UAAU,iBAAiB;;EAE3B;;UAGe;EACf;EACA;EACA;;KAGU,uBAAuB;EAE7B;EACA,gBAAgB;EAChB;EACA,UAAU;;EAEV;EAAW;EAAgB;;;;;;;;;iBASX,uBAAuB,OAC3C,QAAQ,sBAAsB,QAC9B;EAAU;IACT,QAAQ,uBAAuB;UA8BjB;EACf;EACA;EACA;EACA;;EAEA,QAAQ,SAAS;;;;;;;;;;;;iBAaH,oBAAoB,UAAU;;;;iBC/iB9B,kBAAkB,mBAAmB,WAAW;;;;;;iBAShD,uBACd,mBACA,mBACC,cAAc"}
|