@tangle-network/agent-eval 0.150.2 → 0.161.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +158 -1
- package/README.md +7 -3
- package/dist/{active-curriculum-C4mk67HP.js → active-curriculum-CD5TU2yW.js} +3 -13
- package/dist/active-curriculum-CD5TU2yW.js.map +1 -0
- package/dist/{agent-profile-cell-BkcRDikH.d.ts → agent-profile-cell-CTOZJUuE.d.ts} +4 -2
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +19 -36
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/analyst/index.js.map +1 -1
- package/dist/{backend-integrity-DOCa_QrR.d.ts → backend-integrity-DxuQCu_A.d.ts} +4 -3
- package/dist/backend-integrity-DxuQCu_A.d.ts.map +1 -0
- package/dist/{benchmark-BtAWA8nT.d.ts → benchmark-CGPp-kDC.d.ts} +3 -3
- package/dist/{benchmark-BtAWA8nT.d.ts.map → benchmark-CGPp-kDC.d.ts.map} +1 -1
- package/dist/{benchmark-command-BU1Las59.js → benchmark-command-BDC3Gocz.js} +26 -31
- package/dist/benchmark-command-BDC3Gocz.js.map +1 -0
- package/dist/benchmarks/index.d.ts +6 -19
- package/dist/benchmarks/index.d.ts.map +1 -1
- package/dist/benchmarks/index.js +4 -4
- package/dist/benchmarks/index.js.map +1 -1
- package/dist/builder-eval/index.d.ts +3 -3
- package/dist/builder-eval/index.js +2 -2
- package/dist/campaign/index.d.ts +9 -9
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-la-gEYNz.js → campaign-BSmOwskD.js} +77 -795
- package/dist/campaign-BSmOwskD.js.map +1 -0
- package/dist/{canonical-D-XsTQ6_.js → canonical-IL-Bu-14.js} +26 -2
- package/dist/canonical-IL-Bu-14.js.map +1 -0
- package/dist/{capture-fetch-BBVFzhkk.d.ts → capture-fetch-CqwsJkkG.d.ts} +3 -3
- package/dist/{capture-fetch-BBVFzhkk.d.ts.map → capture-fetch-CqwsJkkG.d.ts.map} +1 -1
- package/dist/{chat-client-Bvmxedyv.js → chat-client-DlMlAeYI.js} +5 -55
- package/dist/{chat-client-Bvmxedyv.js.map → chat-client-DlMlAeYI.js.map} +1 -1
- package/dist/chat-json-call-6g5sJobJ.js +53 -0
- package/dist/chat-json-call-6g5sJobJ.js.map +1 -0
- package/dist/cli.js +54 -19
- package/dist/cli.js.map +1 -1
- package/dist/{client-LIuo-KPv.js → client-CX7KqIdB.js} +3 -3
- package/dist/client-CX7KqIdB.js.map +1 -0
- package/dist/{client-kPQYT_56.d.ts → client-L9VVPkim.d.ts} +4 -4
- package/dist/{client-kPQYT_56.d.ts.map → client-L9VVPkim.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -27
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +14 -17
- package/dist/contract/index.js.map +1 -1
- package/dist/{counterfactual--bpysZF0.d.ts → counterfactual-BaFUWK3H.d.ts} +4 -4
- package/dist/{counterfactual--bpysZF0.d.ts.map → counterfactual-BaFUWK3H.d.ts.map} +1 -1
- package/dist/{counterfactual-lDfCx0Uz.js → counterfactual-D_VWavVm.js} +2 -2
- package/dist/{counterfactual-lDfCx0Uz.js.map → counterfactual-D_VWavVm.js.map} +1 -1
- package/dist/{dataset-CJjKqQfA.d.ts → dataset-DQqhOCPt.d.ts} +5 -4
- package/dist/{dataset-CJjKqQfA.d.ts.map → dataset-DQqhOCPt.d.ts.map} +1 -1
- package/dist/{default-registry-Cw0Ohdoj.d.ts → default-registry-G9CKMNkc.d.ts} +7 -8
- package/dist/{default-registry-Cw0Ohdoj.d.ts.map → default-registry-G9CKMNkc.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CEQWL9Hy.d.ts → define-agent-eval-Dx1JnPEa.d.ts} +26 -6
- package/dist/define-agent-eval-Dx1JnPEa.d.ts.map +1 -0
- package/dist/{define-agent-eval-C8V8sMqP.js → define-agent-eval-h-s-sI-v.js} +15 -9
- package/dist/define-agent-eval-h-s-sI-v.js.map +1 -0
- package/dist/{dspy-rlm-engine-CBYlPvNy.js → dspy-rlm-engine-DptEII26.js} +95 -15
- package/dist/dspy-rlm-engine-DptEII26.js.map +1 -0
- package/dist/{emitter-CPBAhxum.js → emitter-BpYFQPj4.js} +2 -18
- package/dist/emitter-BpYFQPj4.js.map +1 -0
- package/dist/{emitter-DGQGoLyj.d.ts → emitter-D_jYSGRd.d.ts} +4 -20
- package/dist/{emitter-DGQGoLyj.d.ts.map → emitter-D_jYSGRd.d.ts.map} +1 -1
- package/dist/{engine-BLzhNzoY.d.ts → engine-Cu5qD5Fc.d.ts} +9 -11
- package/dist/{engine-BLzhNzoY.d.ts.map → engine-Cu5qD5Fc.d.ts.map} +1 -1
- package/dist/{eval-campaign-CQuZrLR_.js → eval-campaign-BsXWL2-2.js} +17 -28
- package/dist/eval-campaign-BsXWL2-2.js.map +1 -0
- package/dist/{exact-types-ccQAyut1.d.ts → exact-types-qnexxJ1Z.d.ts} +2 -2
- package/dist/{exact-types-ccQAyut1.d.ts.map → exact-types-qnexxJ1Z.d.ts.map} +1 -1
- package/dist/{exec-y-DCLqK7.js → exec-D9WpA2p-.js} +2 -2
- package/dist/exec-D9WpA2p-.js.map +1 -0
- package/dist/experiment/index.d.ts +45 -11
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +30 -11
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-0MhuPArU.d.ts → experiment-tracker-DCO6Cz4s.d.ts} +2 -2
- package/dist/{experiment-tracker-0MhuPArU.d.ts.map → experiment-tracker-DCO6Cz4s.d.ts.map} +1 -1
- package/dist/{exporters-q9iL-2Jf.js → exporters-Df7TgHFv.js} +3 -3
- package/dist/exporters-Df7TgHFv.js.map +1 -0
- package/dist/{external-optimizer-contracts-DbLsm4Po.d.ts → external-optimizer-contracts-szBJ_1vh.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-DbLsm4Po.d.ts.map → external-optimizer-contracts-szBJ_1vh.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-x9oEXKsU.js → external-optimizer-process-WosTBChy.js} +4 -4
- package/dist/{external-optimizer-process-x9oEXKsU.js.map → external-optimizer-process-WosTBChy.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-CKNb42oM.js → external-optimizer-subprocess-BIWbHpgD.js} +4 -4
- package/dist/{external-optimizer-subprocess-CKNb42oM.js.map → external-optimizer-subprocess-BIWbHpgD.js.map} +1 -1
- package/dist/{failure-cluster-BLURuWG4.d.ts → failure-cluster-CXL8NbEw.d.ts} +3 -3
- package/dist/{failure-cluster-BLURuWG4.d.ts.map → failure-cluster-CXL8NbEw.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-DpTTjo0q.d.ts → feedback-trajectory-B3ZHaHV_.d.ts} +7 -7
- package/dist/{feedback-trajectory-DpTTjo0q.d.ts.map → feedback-trajectory-B3ZHaHV_.d.ts.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/{hf-dataset-XggBupCr.js → hf-dataset-D8_RNIis.js} +4 -4
- package/dist/{hf-dataset-XggBupCr.js.map → hf-dataset-D8_RNIis.js.map} +1 -1
- package/dist/hosted/index.d.ts +3 -14
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +2 -2
- package/dist/{index-Aj3WO3_a.d.ts → index-CGtH1piv.d.ts} +48 -26
- package/dist/index-CGtH1piv.d.ts.map +1 -0
- package/dist/{index-BNPtkBPf.d.ts → index-D-IiQIBB.d.ts} +5 -10
- package/dist/index-D-IiQIBB.d.ts.map +1 -0
- package/dist/{index-IQccV3Ou.d.ts → index-D-V8gCs_.d.ts} +13 -90
- package/dist/index-D-V8gCs_.d.ts.map +1 -0
- package/dist/{index-B8Ui1mr1.d.ts → index-lfaSeKSD.d.ts} +18 -2
- package/dist/index-lfaSeKSD.d.ts.map +1 -0
- package/dist/index-vrJugRal.d.ts +1 -0
- package/dist/index.d.ts +68 -56
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +59 -110
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-BeT8KCgI.d.ts → insight-report-DRe8LB6d.d.ts} +4 -4
- package/dist/{insight-report-BeT8KCgI.d.ts.map → insight-report-DRe8LB6d.d.ts.map} +1 -1
- package/dist/{integrity-DL91tucI.js → integrity-CyWSSoQS.js} +2 -2
- package/dist/{integrity-DL91tucI.js.map → integrity-CyWSSoQS.js.map} +1 -1
- package/dist/{integrity-B0dZ96EO.d.ts → integrity-DUNX9Fao.d.ts} +3 -3
- package/dist/{integrity-B0dZ96EO.d.ts.map → integrity-DUNX9Fao.d.ts.map} +1 -1
- package/dist/{kind-factory-DmAa0h3K.js → kind-factory-DY8FdoXf.js} +3 -71
- package/dist/kind-factory-DY8FdoXf.js.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +3 -3
- package/dist/{ledger-core-DTae9rv_.js → ledger-core-BOzlRygb.js} +2 -2
- package/dist/{ledger-core-DTae9rv_.js.map → ledger-core-BOzlRygb.js.map} +1 -1
- package/dist/{llm-client-Bg32RW0j.js → llm-client-hgDieDNN.js} +53 -98
- package/dist/llm-client-hgDieDNN.js.map +1 -0
- package/dist/{llm-judge-BqqMS8t7.js → llm-judge-BhasIPFT.js} +1135 -77
- package/dist/llm-judge-BhasIPFT.js.map +1 -0
- package/dist/{matrix-DrVnRp4G.d.ts → matrix-eXKRMHnL.d.ts} +74 -72
- package/dist/matrix-eXKRMHnL.d.ts.map +1 -0
- package/dist/meta-eval/index.d.ts +8 -6
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +8 -6
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-BV6tLVWl.js → mint-DfODW1KW.js} +3 -3
- package/dist/{mint-BV6tLVWl.js.map → mint-DfODW1KW.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +2 -8
- package/dist/multishot/golden/index.d.ts.map +1 -1
- package/dist/multishot/golden/index.js +56 -86
- package/dist/multishot/golden/index.js.map +1 -1
- package/dist/multishot/index.d.ts +11 -46
- package/dist/multishot/index.d.ts.map +1 -1
- package/dist/multishot/index.js +30 -83
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{opencode-sqlite-DJWAXLms.js → opencode-sqlite-eK6HW6dr.js} +2 -6
- package/dist/{opencode-sqlite-DJWAXLms.js.map → opencode-sqlite-eK6HW6dr.js.map} +1 -1
- package/dist/pipelines/index.d.ts +5 -5
- package/dist/pipelines/index.js +3 -3
- package/dist/{pre-registration-zFSLEiFU.d.ts → pre-registration-CzFCcwYk.d.ts} +55 -40
- package/dist/pre-registration-CzFCcwYk.d.ts.map +1 -0
- package/dist/pre-registration-KN9jkh58.js +110 -0
- package/dist/pre-registration-KN9jkh58.js.map +1 -0
- package/dist/{produced-state-Bnq4FaDO.js → produced-state-DZ89riy5.js} +8 -8
- package/dist/produced-state-DZ89riy5.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +31 -5
- package/dist/profile-cell.js.map +1 -1
- package/dist/{promotion-policy-DLOUkYhI.d.ts → promotion-policy-DtnOIZvk.d.ts} +2 -2
- package/dist/{promotion-policy-DLOUkYhI.d.ts.map → promotion-policy-DtnOIZvk.d.ts.map} +1 -1
- package/dist/{query-Di7eEQ79.js → query-CHmMP42p.js} +20 -11
- package/dist/query-CHmMP42p.js.map +1 -0
- package/dist/{query-CJ_DX8vl.d.ts → query-DxPYqpmT.d.ts} +10 -4
- package/dist/query-DxPYqpmT.d.ts.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/{registry-BQwrSYpC.d.ts → registry-8You7OK1.d.ts} +5 -7
- package/dist/{registry-BQwrSYpC.d.ts.map → registry-8You7OK1.d.ts.map} +1 -1
- package/dist/{release-confidence-BknrpBnO.js → release-confidence-DKfD2RYU.js} +28 -14
- package/dist/release-confidence-DKfD2RYU.js.map +1 -0
- package/dist/{release-confidence-4XrqlpFD.d.ts → release-confidence-Dqt0NFep.d.ts} +7 -6
- package/dist/release-confidence-Dqt0NFep.d.ts.map +1 -0
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +3 -3
- package/dist/{researcher-DJnoUE8c.d.ts → researcher-Cz565b7D.d.ts} +34 -21
- package/dist/researcher-Cz565b7D.d.ts.map +1 -0
- package/dist/{reward-hacking-62tojkQd.d.ts → reward-hacking-MBf7qpSB.d.ts} +2 -2
- package/dist/{reward-hacking-62tojkQd.d.ts.map → reward-hacking-MBf7qpSB.d.ts.map} +1 -1
- package/dist/{reward-hacking-C0x0xihA.js → reward-hacking-t4lB1yt8.js} +2 -2
- package/dist/{reward-hacking-C0x0xihA.js.map → reward-hacking-t4lB1yt8.js.map} +1 -1
- package/dist/rl.d.ts +11 -42
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +41 -24
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +3 -3
- package/dist/rollout/index.js +7 -7
- package/dist/{rollout-ytVQ7WT8.js → rollout-Dm2tSdiQ.js} +6 -6
- package/dist/{rollout-ytVQ7WT8.js.map → rollout-Dm2tSdiQ.js.map} +1 -1
- package/dist/{rubric-predictive-validity-CzxLoZge.js → rubric-predictive-validity-CK8SCOg-.js} +5 -15
- package/dist/rubric-predictive-validity-CK8SCOg-.js.map +1 -0
- package/dist/{rubric-predictive-validity-C7LnNvF2.d.ts → rubric-predictive-validity-CxycqzX5.d.ts} +4 -3
- package/dist/rubric-predictive-validity-CxycqzX5.d.ts.map +1 -0
- package/dist/{run-record-D2lDdSAz.js → run-record-BC0ebuRP.js} +2 -2
- package/dist/{run-record-D2lDdSAz.js.map → run-record-BC0ebuRP.js.map} +1 -1
- package/dist/{run-record-DVV82Gwh.d.ts → run-record-VVy4T9OW.d.ts} +3 -3
- package/dist/{run-record-DVV82Gwh.d.ts.map → run-record-VVy4T9OW.d.ts.map} +1 -1
- package/dist/{schema-BtVldJ3T.d.ts → schema-Bjgdsn73.d.ts} +2 -4
- package/dist/{schema-BtVldJ3T.d.ts.map → schema-Bjgdsn73.d.ts.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-BzWDXhOR.d.ts} +2 -5
- package/dist/schema-BzWDXhOR.d.ts.map +1 -0
- package/dist/{schema-C6DW4ZHR.js → schema-C1aaAxTf.js} +2 -2
- package/dist/schema-C1aaAxTf.js.map +1 -0
- package/dist/{schema-CRhEY1SO.js → schema-k6ZBftVv.js} +2 -8
- package/dist/{schema-CRhEY1SO.js.map → schema-k6ZBftVv.js.map} +1 -1
- package/dist/{semantic-concept-judge-laMCnTLn.js → semantic-concept-judge-BSkKKHeq.js} +14 -38
- package/dist/semantic-concept-judge-BSkKKHeq.js.map +1 -0
- package/dist/{sequential-eprocess-CbUt2htw.js → sequential-eprocess-D1jKoihe.js} +49 -2
- package/dist/sequential-eprocess-D1jKoihe.js.map +1 -0
- package/dist/{sequential-C458DXNf.js → sequential-rYW-Ophm.js} +41 -16
- package/dist/sequential-rYW-Ophm.js.map +1 -0
- package/dist/{series-convergence-BxKEgBwA.d.ts → series-convergence-D9WgpXGi.d.ts} +2 -2
- package/dist/{series-convergence-BxKEgBwA.d.ts.map → series-convergence-D9WgpXGi.d.ts.map} +1 -1
- package/dist/{server-dIWwF3j_.js → server-BtFd4uzB.js} +19 -42
- package/dist/server-BtFd4uzB.js.map +1 -0
- package/dist/{skillopt-optimization-method-CPBlTcj5.js → skillopt-optimization-method-DbaekMcn.js} +794 -8
- package/dist/skillopt-optimization-method-DbaekMcn.js.map +1 -0
- package/dist/{skillopt-optimization-method-BDD_o1xE.d.ts → skillopt-optimization-method-x7TTF23P.d.ts} +20 -7
- package/dist/{skillopt-optimization-method-BDD_o1xE.d.ts.map → skillopt-optimization-method-x7TTF23P.d.ts.map} +1 -1
- package/dist/{statistical-heldout-_woZ9q9j.d.ts → statistical-heldout-Cy3EhjlC.d.ts} +21 -8
- package/dist/statistical-heldout-Cy3EhjlC.d.ts.map +1 -0
- package/dist/{steps-AmkT-GIM.d.ts → steps-CiNVJry_.d.ts} +2 -17
- package/dist/steps-CiNVJry_.d.ts.map +1 -0
- package/dist/{store-CT9YIIve.d.ts → store-B06JdC56.d.ts} +2 -2
- package/dist/{store-CT9YIIve.d.ts.map → store-B06JdC56.d.ts.map} +1 -1
- package/dist/{store-otlp-CDYWW_8N.js → store-otlp-C_Rq5I4D.js} +2 -2
- package/dist/{store-otlp-CDYWW_8N.js.map → store-otlp-C_Rq5I4D.js.map} +1 -1
- package/dist/{store-tool-spans-Br2_IUhm.d.ts → store-tool-spans-DPUG7UUY.d.ts} +6 -6
- package/dist/{store-tool-spans-Br2_IUhm.d.ts.map → store-tool-spans-DPUG7UUY.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CykkbOlv.js → store-tool-spans-Dlh9vkFK.js} +3 -3
- package/dist/{store-tool-spans-CykkbOlv.js.map → store-tool-spans-Dlh9vkFK.js.map} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-DW2bEpdB.js → summary-report-BI5hUtvK.js} +6 -16
- package/dist/summary-report-BI5hUtvK.js.map +1 -0
- package/dist/{summary-report-B__Y5ub3.d.ts → summary-report-CC07PhEL.d.ts} +6 -5
- package/dist/summary-report-CC07PhEL.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +3 -16
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +3 -3
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{task-failure-attributes--ZTP3tYO.js → task-failure-attributes-DTl-7-Kw.js} +3 -3
- package/dist/{task-failure-attributes--ZTP3tYO.js.map → task-failure-attributes-DTl-7-Kw.js.map} +1 -1
- package/dist/{tool-groups-B4tqh8jB.d.ts → tool-groups-Ci8i9ErB.d.ts} +3 -3
- package/dist/tool-groups-Ci8i9ErB.d.ts.map +1 -0
- package/dist/{tool-waste-C-VHSRwF.js → tool-waste-BqzmVdJk.js} +2 -2
- package/dist/{tool-waste-C-VHSRwF.js.map → tool-waste-BqzmVdJk.js.map} +1 -1
- package/dist/{tool-waste-DjRDEsuI.d.ts → tool-waste-Dro0gJi3.d.ts} +4 -4
- package/dist/{tool-waste-DjRDEsuI.d.ts.map → tool-waste-Dro0gJi3.d.ts.map} +1 -1
- package/dist/trace-repair/index.d.ts +4 -77
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +5 -15
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +13 -23
- package/dist/traces.d.ts.map +1 -1
- package/dist/traces.js +9 -20
- package/dist/traces.js.map +1 -1
- package/dist/{trajectory-YC15QDYQ.d.ts → trajectory-Bi157Gun.d.ts} +3 -3
- package/dist/{trajectory-YC15QDYQ.d.ts.map → trajectory-Bi157Gun.d.ts.map} +1 -1
- package/dist/trajectory-replay/index.d.ts +5 -5
- package/dist/trajectory-replay/index.js +5 -5
- package/dist/{provenance-oA4-zUqm.d.ts → transient-failure-DKF5Mofa.d.ts} +468 -13
- package/dist/transient-failure-DKF5Mofa.d.ts.map +1 -0
- package/dist/types-B3jzCp0p.js.map +1 -1
- package/dist/{types-CLAwnY-L.d.ts → types-BPb2Kf_C2.d.ts} +3 -3
- package/dist/types-BPb2Kf_C2.d.ts.map +1 -0
- package/dist/types-Bfk0uxRj.d.ts +443 -0
- package/dist/types-Bfk0uxRj.d.ts.map +1 -0
- package/dist/{types-DdFNuyxQ.d.ts → types-D4s7Z6nq.d.ts} +30 -6
- package/dist/types-D4s7Z6nq.d.ts.map +1 -0
- package/dist/{types-B2NsbrNy.d.ts → types-D9ssmxKL.d.ts} +3 -3
- package/dist/{types-B2NsbrNy.d.ts.map → types-D9ssmxKL.d.ts.map} +1 -1
- package/dist/{types-I5WwVzQ7.d.ts → types-DeIUdzNd.d.ts} +2 -2
- package/dist/{types-I5WwVzQ7.d.ts.map → types-DeIUdzNd.d.ts.map} +1 -1
- package/dist/{verdict-BndeTAh_.js → verdict-B0xltqu6.js} +2 -2
- package/dist/{verdict-BndeTAh_.js.map → verdict-B0xltqu6.js.map} +1 -1
- package/dist/verdict-cache-CdVVTVmn.js +88 -0
- package/dist/verdict-cache-CdVVTVmn.js.map +1 -0
- package/dist/wire/index.d.ts +21 -111
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +2 -2
- package/docs/building-doctrine.md +3 -3
- package/docs/campaign-proposers.md +41 -10
- package/docs/design/statistics-decisions.md +89 -1
- package/docs/eval-surface-map.md +14 -0
- package/docs/experiment.md +19 -2
- package/docs/feedback-trajectories.md +1 -1
- package/docs/multishot-golden-records.md +4 -4
- package/docs/public-api.md +1616 -0
- package/docs/research-report-methodology.md +1 -1
- package/docs/search-history-receipts.md +39 -1
- package/docs/trace-analysis.md +1 -1
- package/docs/trace-repair-admission.md +1 -1
- package/docs/trace-repair-continuation.md +1 -1
- package/docs/verdicts.md +24 -0
- package/package.json +7 -3
- package/dist/active-curriculum-C4mk67HP.js.map +0 -1
- package/dist/agent-profile-cell-BkcRDikH.d.ts.map +0 -1
- package/dist/backend-integrity-DOCa_QrR.d.ts.map +0 -1
- package/dist/benchmark-command-BU1Las59.js.map +0 -1
- package/dist/campaign-la-gEYNz.js.map +0 -1
- package/dist/canonical-D-XsTQ6_.js.map +0 -1
- package/dist/client-LIuo-KPv.js.map +0 -1
- package/dist/define-agent-eval-C8V8sMqP.js.map +0 -1
- package/dist/define-agent-eval-CEQWL9Hy.d.ts.map +0 -1
- package/dist/dspy-rlm-engine-CBYlPvNy.js.map +0 -1
- package/dist/emitter-CPBAhxum.js.map +0 -1
- package/dist/eval-campaign-CQuZrLR_.js.map +0 -1
- package/dist/exec-y-DCLqK7.js.map +0 -1
- package/dist/exporters-q9iL-2Jf.js.map +0 -1
- package/dist/index-Aj3WO3_a.d.ts.map +0 -1
- package/dist/index-B8Ui1mr1.d.ts.map +0 -1
- package/dist/index-BNPtkBPf.d.ts.map +0 -1
- package/dist/index-C1ravkGA.d.ts +0 -1
- package/dist/index-IQccV3Ou.d.ts.map +0 -1
- package/dist/kind-factory-DmAa0h3K.js.map +0 -1
- package/dist/llm-client-Bg32RW0j.js.map +0 -1
- package/dist/llm-judge-BqqMS8t7.js.map +0 -1
- package/dist/matrix-DrVnRp4G.d.ts.map +0 -1
- package/dist/pre-registration-DakwTRXk.js +0 -96
- package/dist/pre-registration-DakwTRXk.js.map +0 -1
- package/dist/pre-registration-zFSLEiFU.d.ts.map +0 -1
- package/dist/produced-state-Bnq4FaDO.js.map +0 -1
- package/dist/provenance-oA4-zUqm.d.ts.map +0 -1
- package/dist/query-CJ_DX8vl.d.ts.map +0 -1
- package/dist/query-Di7eEQ79.js.map +0 -1
- package/dist/release-confidence-4XrqlpFD.d.ts.map +0 -1
- package/dist/release-confidence-BknrpBnO.js.map +0 -1
- package/dist/researcher-DJnoUE8c.d.ts.map +0 -1
- package/dist/rubric-predictive-validity-C7LnNvF2.d.ts.map +0 -1
- package/dist/rubric-predictive-validity-CzxLoZge.js.map +0 -1
- package/dist/schema-C6DW4ZHR.js.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/semantic-concept-judge-laMCnTLn.js.map +0 -1
- package/dist/sequential-C458DXNf.js.map +0 -1
- package/dist/sequential-eprocess-CbUt2htw.js.map +0 -1
- package/dist/server-dIWwF3j_.js.map +0 -1
- package/dist/skillopt-optimization-method-CPBlTcj5.js.map +0 -1
- package/dist/statistical-heldout-_woZ9q9j.d.ts.map +0 -1
- package/dist/steps-AmkT-GIM.d.ts.map +0 -1
- package/dist/summary-report-B__Y5ub3.d.ts.map +0 -1
- package/dist/summary-report-DW2bEpdB.js.map +0 -1
- package/dist/tool-groups-B4tqh8jB.d.ts.map +0 -1
- package/dist/types-CLAwnY-L.d.ts.map +0 -1
- package/dist/types-DdFNuyxQ.d.ts.map +0 -1
- package/dist/types-jUBXJ7Iz.d.ts +0 -884
- package/dist/types-jUBXJ7Iz.d.ts.map +0 -1
- package/dist/verdict-cache-mZf5FEiY.js +0 -107
- package/dist/verdict-cache-mZf5FEiY.js.map +0 -1
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { t as DefaultVerdict } from "../verdict-E4eRNf7-.js";
|
|
2
|
-
import { s as TraceStore } from "../store-
|
|
3
|
-
import { i as TraceEmitter } from "../emitter-
|
|
4
|
-
import { i as CounterfactualRunner, t as CounterfactualContext } from "../counterfactual
|
|
5
|
-
import { _ as
|
|
2
|
+
import { s as TraceStore } from "../store-B06JdC56.js";
|
|
3
|
+
import { i as TraceEmitter } from "../emitter-D_jYSGRd.js";
|
|
4
|
+
import { i as CounterfactualRunner, t as CounterfactualContext } from "../counterfactual-BaFUWK3H.js";
|
|
5
|
+
import { _ as isSubmitOnlyAction, a as RecordedTrajectoryStep, b as unreadableExitCount, c as TIMEOUT_OBSERVATION_MARKER, d as decodeRecordedTurns, f as deriveFailureSignature, g as isSubmitAction, h as isRecordedTimeout, i as RecordedObservationKind, l as assertReplayableTrajectory, m as isElidedField, n as FORMAT_ERROR_OBSERVATION_PREFIX, o as RecordedTrajectoryTurn, p as finalRecordedOutcome, r as RecordedFinalOutcome, s as SUBMIT_ACTION_SIGNATURE, t as DecodedTrajectory, u as classifyObservation, v as parseObservationOutput, y as parseRecordedReturncode } from "../steps-CiNVJry_.js";
|
|
6
6
|
//#region src/trajectory-replay/corpus.d.ts
|
|
7
7
|
interface CorpusSpec {
|
|
8
8
|
readonly name: string;
|
|
@@ -790,5 +790,5 @@ interface ReplayFindingResult {
|
|
|
790
790
|
*/
|
|
791
791
|
declare function replayVerifyFinding(finding: AnalystReplayFinding, options: ReplayFindingOptions): Promise<ReplayFindingResult>;
|
|
792
792
|
//#endregion
|
|
793
|
-
export { type AnalystReplayFinding, type ArmExecutionResult, type CaseResources, type ChatCompletionCaller, type ChatOutcome, type ChatUsage, type CorpusSpec, type CwdSource, type DecodedTrajectory, type DockerImagePreparerOptions, type EnumerationResult, type ExcludedCase, FORMAT_ERROR_OBSERVATION_PREFIX, type FailedFixAttempt, type FindingReplaySource, type FindingReplayability, type FindingVerification, type FindingVerificationStatus, type FixArmExecution, type FixArmExecutor, type FixGenerationOutcome, type FixLoopAttemptRecord, type FixLoopOptions, type FixLoopResult, type FixPromptInput, type ImagePreparation, type ImagePreparer, type IncorrectStepsSubject, type IngestedTrajectory, PREFIX_DIVERGENCE_TOLERANCE_PCT, type PrefixDivergence, type PrefixDivergenceKind, type PrefixReplayResult,
|
|
793
|
+
export { type AnalystReplayFinding, type ArmExecutionResult, type CaseResources, type ChatCompletionCaller, type ChatOutcome, type ChatUsage, type CorpusSpec, type CwdSource, type DecodedTrajectory, type DockerImagePreparerOptions, type EnumerationResult, type ExcludedCase, FORMAT_ERROR_OBSERVATION_PREFIX, type FailedFixAttempt, type FindingReplaySource, type FindingReplayability, type FindingVerification, type FindingVerificationStatus, type FixArmExecution, type FixArmExecutor, type FixGenerationOutcome, type FixLoopAttemptRecord, type FixLoopOptions, type FixLoopResult, type FixPromptInput, type ImagePreparation, type ImagePreparer, type IncorrectStepsSubject, type IngestedTrajectory, PREFIX_DIVERGENCE_TOLERANCE_PCT, type PrefixDivergence, type PrefixDivergenceKind, type PrefixReplayResult, type RecordedFinalOutcome, type RecordedObservationKind, type RecordedTrajectoryStep, type RecordedTrajectoryTurn, type ReplayArmVerdict, type ReplayBatchCaseRow, type ReplayBatchFixResult, type ReplayBatchOptions, type ReplayBatchReport, type ReplayExclusionReason, type ReplayExecBackend, type ReplayExecBackendFactory, type ReplayExecResult, type ReplayExecSession, type ReplayFindingOptions, type ReplayFindingResult, type ReplayVerdict, type ReplayVerifyOptions, type ReplayableCase, type ResolvedFindingReplay, type ResolvedReplayInvocation, type ResourceResolution, SUBMIT_ACTION_SIGNATURE, SandboxCounterfactualRunner, type SandboxCounterfactualRunnerOptions, TIMEOUT_OBSERVATION_MARKER, type VerifiableFinding, type VerifyFindingsOptions, type VerifyFindingsRun, assertReplayableTrajectory, buildFixPrompt, buildRetryFixPrompt, classifyObservation, classifyPrefixStep, classifyVerdict, clipText, countScriptCommands, decodeRecordedTurns, deriveFailureSignature, derivedImageTag, dockerImagePreparer, enumerateReplayableCases, extractFixCommand, finalRecordedOutcome, findingReplayStep, findingTrajectoryId, generateFixCommand, goldIncorrectSteps, ingestRecordedTrajectory, isElidedField, isRecordedTimeout, isSubmitAction, isSubmitOnlyAction, parseCorpusFlag, parseIncorrectStepsSubject, parseObservationOutput, parseRecordedReturncode, readFindingsFile, readLabelEntries, renderBatchReport, renderVerifiedFindingsSection, replayVerify, replayVerifyFinding, resolveCaseResources, resolveFindingInvocation, resolveFindingReplayability, runFixLoop, runReplayBatch, seededSample, summarizePrefixReplay, unreadableExitCount, verifyFindings, wrapActionForExec };
|
|
794
794
|
//# sourceMappingURL=index.d.ts.map
|
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
import { t as certificationEvidenceDigest } from "../verdict-
|
|
2
|
-
import { t as TraceEmitter } from "../emitter-
|
|
3
|
-
import { t as runCounterfactual } from "../counterfactual-
|
|
1
|
+
import { t as certificationEvidenceDigest } from "../verdict-B0xltqu6.js";
|
|
2
|
+
import { t as TraceEmitter } from "../emitter-BpYFQPj4.js";
|
|
3
|
+
import { t as runCounterfactual } from "../counterfactual-D_VWavVm.js";
|
|
4
4
|
import { n as InMemoryTraceStore } from "../store-DNe_Uv1Q.js";
|
|
5
5
|
import { t as packageVersion } from "../package-version-D7lQHt_-.js";
|
|
6
|
-
import {
|
|
6
|
+
import { a as assertReplayableTrajectory, c as deriveFailureSignature, d as isRecordedTimeout, f as isSubmitAction, g as unreadableExitCount, h as parseRecordedReturncode, i as TIMEOUT_OBSERVATION_MARKER, l as finalRecordedOutcome, m as parseObservationOutput, n as FORMAT_ERROR_OBSERVATION_PREFIX, o as classifyObservation, p as isSubmitOnlyAction, r as SUBMIT_ACTION_SIGNATURE, s as decodeRecordedTurns, t as wrapActionForExec, u as isElidedField } from "../exec-D9WpA2p-.js";
|
|
7
7
|
import { createHash } from "node:crypto";
|
|
8
8
|
import { appendFileSync, existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, writeFileSync } from "node:fs";
|
|
9
9
|
import { join } from "node:path";
|
|
@@ -2145,6 +2145,6 @@ function readFindingsFile(path) {
|
|
|
2145
2145
|
return array;
|
|
2146
2146
|
}
|
|
2147
2147
|
//#endregion
|
|
2148
|
-
export { FORMAT_ERROR_OBSERVATION_PREFIX, PREFIX_DIVERGENCE_TOLERANCE_PCT,
|
|
2148
|
+
export { FORMAT_ERROR_OBSERVATION_PREFIX, PREFIX_DIVERGENCE_TOLERANCE_PCT, SUBMIT_ACTION_SIGNATURE, SandboxCounterfactualRunner, TIMEOUT_OBSERVATION_MARKER, assertReplayableTrajectory, buildFixPrompt, buildRetryFixPrompt, classifyObservation, classifyPrefixStep, classifyVerdict, clipText, countScriptCommands, decodeRecordedTurns, deriveFailureSignature, derivedImageTag, dockerImagePreparer, enumerateReplayableCases, extractFixCommand, finalRecordedOutcome, findingReplayStep, findingTrajectoryId, generateFixCommand, goldIncorrectSteps, ingestRecordedTrajectory, isElidedField, isRecordedTimeout, isSubmitAction, isSubmitOnlyAction, parseCorpusFlag, parseIncorrectStepsSubject, parseObservationOutput, parseRecordedReturncode, readFindingsFile, readLabelEntries, renderBatchReport, renderVerifiedFindingsSection, replayVerify, replayVerifyFinding, resolveCaseResources, resolveFindingInvocation, resolveFindingReplayability, runFixLoop, runReplayBatch, seededSample, summarizePrefixReplay, unreadableExitCount, verifyFindings, wrapActionForExec };
|
|
2149
2149
|
|
|
2150
2150
|
//# sourceMappingURL=index.js.map
|
|
@@ -1,13 +1,169 @@
|
|
|
1
|
+
import { t as AgentEvalError } from "./errors-DEE6u6ot.js";
|
|
1
2
|
import { c as CostLedgerHandle, f as CostLedgerSummary, m as CostReceipt, o as CostLedger, p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
|
|
2
|
-
import { a as RunRecord } from "./run-record-
|
|
3
|
-
import { p as ChatClient } from "./types-
|
|
4
|
-
import { _ as ProposalFinding } from "./types-
|
|
5
|
-
import { _ as ExternalTextCandidate, g as ExternalOptimizerWireCounts, o as ExternalOptimizerEvaluationObservation } from "./external-optimizer-contracts-
|
|
6
|
-
import { C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, j as MutableSurface, k as LabeledScenarioStore, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-
|
|
7
|
-
import { r as DatasetScenario, t as Dataset } from "./dataset-
|
|
8
|
-
import { D as LedgerHash, m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-
|
|
9
|
-
import { g as TraceSpanEvent, t as HostedClient } from "./client-
|
|
3
|
+
import { a as RunRecord } from "./run-record-VVy4T9OW.js";
|
|
4
|
+
import { p as ChatClient } from "./types-Bfk0uxRj.js";
|
|
5
|
+
import { _ as ProposalFinding } from "./types-D9ssmxKL.js";
|
|
6
|
+
import { _ as ExternalTextCandidate, g as ExternalOptimizerWireCounts, o as ExternalOptimizerEvaluationObservation } from "./external-optimizer-contracts-szBJ_1vh.js";
|
|
7
|
+
import { B as ScoredSurfaceOutcome, C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, j as MutableSurface, k as LabeledScenarioStore, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-D4s7Z6nq.js";
|
|
8
|
+
import { r as DatasetScenario, t as Dataset } from "./dataset-DQqhOCPt.js";
|
|
9
|
+
import { D as LedgerHash, m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-lfaSeKSD.js";
|
|
10
|
+
import { g as TraceSpanEvent, t as HostedClient } from "./client-L9VVPkim.js";
|
|
10
11
|
import { z } from "zod";
|
|
12
|
+
//#region src/judge-families.d.ts
|
|
13
|
+
/**
|
|
14
|
+
* Judge model-family classification + cross-family enforcement.
|
|
15
|
+
*
|
|
16
|
+
* A judge ensemble built entirely from one provider family shares that
|
|
17
|
+
* family's blind spots and self-preference — its "agreement" is correlated
|
|
18
|
+
* bias, not independent signal. `assertCrossFamily` makes the consumer prove
|
|
19
|
+
* the ensemble spans ≥2 families; `judgeFamily` is the single regex map that
|
|
20
|
+
* replaces the per-consumer copies (tax/legal/creative/gtm each ship one).
|
|
21
|
+
*/
|
|
22
|
+
/** Provider family a model belongs to. `unknown` when no rule matches. */
|
|
23
|
+
type JudgeFamily = 'anthropic' | 'openai' | 'google' | 'meta' | 'mistral' | 'deepseek' | 'xai' | 'qwen' | 'cohere' | 'amazon' | 'moonshot' | 'zhipu' | 'unknown';
|
|
24
|
+
/**
|
|
25
|
+
* Classify a model id into its provider family. Strips a `@snapshot` suffix
|
|
26
|
+
* and prefers an explicit `provider/...` prefix; otherwise matches the model
|
|
27
|
+
* name. Returns `unknown` when nothing matches (callers decide whether that's
|
|
28
|
+
* acceptable — `assertCrossFamily` counts it as its own family).
|
|
29
|
+
*/
|
|
30
|
+
declare function judgeFamily(modelId: string): JudgeFamily;
|
|
31
|
+
interface AssertCrossFamilyOptions {
|
|
32
|
+
/** Minimum number of distinct families the ensemble must span. Default 2. */
|
|
33
|
+
minFamilies?: number;
|
|
34
|
+
/** When false (default), `unknown`-family models do NOT count toward the
|
|
35
|
+
* family total — an ensemble of all-unclassifiable models is not provably
|
|
36
|
+
* cross-family. Set true to count `unknown` as one shared family. */
|
|
37
|
+
allowUnknown?: boolean;
|
|
38
|
+
}
|
|
39
|
+
declare class CrossFamilyError extends Error {
|
|
40
|
+
readonly families: JudgeFamily[];
|
|
41
|
+
readonly models: string[];
|
|
42
|
+
constructor(message: string, families: JudgeFamily[], models: string[]);
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* Throw unless the judge models span at least `minFamilies` distinct provider
|
|
46
|
+
* families. Pass the model ids backing your judge ensemble. Fail-loud by
|
|
47
|
+
* design — a correlated single-family ensemble silently inflates agreement.
|
|
48
|
+
*
|
|
49
|
+
* Scope: this reads the ids you REQUEST. It proves the panel was configured
|
|
50
|
+
* across families; it cannot prove the panel RAN across families, because a
|
|
51
|
+
* routing gateway may answer several different ids from one provider. Where
|
|
52
|
+
* the diversity claim is load-bearing (a published leaderboard, a
|
|
53
|
+
* certification, a non-self-judging exclusion), assert on the ids the
|
|
54
|
+
* provider echoed instead: `assertCrossFamilyServed` in
|
|
55
|
+
* ./integrity/served-model.
|
|
56
|
+
*/
|
|
57
|
+
declare function assertCrossFamily(models: string[], opts?: AssertCrossFamilyOptions): JudgeFamily[];
|
|
58
|
+
//#endregion
|
|
59
|
+
//#region src/integrity/served-model.d.ts
|
|
60
|
+
/** How a served id relates to the id that was requested. */
|
|
61
|
+
type ServedModelVerdict =
|
|
62
|
+
/** Byte-identical after normalisation — the requested model answered. */
|
|
63
|
+
'exact' |
|
|
64
|
+
/** Same model, different spelling (provider prefix, snapshot, tier suffix). */
|
|
65
|
+
'alias' |
|
|
66
|
+
/** A different model of the SAME provider family answered. */
|
|
67
|
+
'substituted-within-family' |
|
|
68
|
+
/** A different provider's model answered. */
|
|
69
|
+
'substituted-cross-family' |
|
|
70
|
+
/** The response carried no model id — identity is unproven either way. */
|
|
71
|
+
'unreported';
|
|
72
|
+
interface ServedModelCheck {
|
|
73
|
+
/** The id the caller asked for. */
|
|
74
|
+
requested: string;
|
|
75
|
+
/** The id echoed on the response; `null` when the response omitted it. */
|
|
76
|
+
served: string | null;
|
|
77
|
+
requestedFamily: JudgeFamily;
|
|
78
|
+
/** `null` when `served` is null. */
|
|
79
|
+
servedFamily: JudgeFamily | null;
|
|
80
|
+
verdict: ServedModelVerdict;
|
|
81
|
+
/** True for every verdict except `exact` and `alias`. */
|
|
82
|
+
substituted: boolean;
|
|
83
|
+
}
|
|
84
|
+
/**
|
|
85
|
+
* Classify one requested/served pair. Pure — no I/O — so it is safe inside
|
|
86
|
+
* response handlers, reducers, and CI gates.
|
|
87
|
+
*
|
|
88
|
+
* `served` is the id echoed by the provider (OpenAI-compatible bodies put it
|
|
89
|
+
* at `model`). `null`/`undefined` means the body omitted it; that is
|
|
90
|
+
* `unreported`, NOT a pass — a provider that does not name what answered has
|
|
91
|
+
* not proven identity, and a transport that drops the field must not read as
|
|
92
|
+
* agreement.
|
|
93
|
+
*/
|
|
94
|
+
declare function checkServedModel(requested: string, served: string | null | undefined): ServedModelCheck;
|
|
95
|
+
declare class ModelSubstitutionError extends AgentEvalError {
|
|
96
|
+
readonly checks: ReadonlyArray<ServedModelCheck>;
|
|
97
|
+
constructor(message: string, checks: ReadonlyArray<ServedModelCheck>);
|
|
98
|
+
}
|
|
99
|
+
/**
|
|
100
|
+
* Consumer-facing name for the substitution policy a metered surface applies.
|
|
101
|
+
* `'exact'` rejects every substitution. `'allow-within-family'` accepts a
|
|
102
|
+
* different model of the same provider family; it keeps family-level claims
|
|
103
|
+
* valid and forfeits per-model claims. Maps to
|
|
104
|
+
* `AssertServedModelOptions.allowWithinFamily`.
|
|
105
|
+
*/
|
|
106
|
+
type ServedModelPolicy = 'exact' | 'allow-within-family';
|
|
107
|
+
interface AssertServedModelOptions {
|
|
108
|
+
/**
|
|
109
|
+
* Accept a different model of the same provider family (e.g. requested
|
|
110
|
+
* `deepseek-v3.2`, served `deepseek-v4-flash`). Default false. Setting this
|
|
111
|
+
* keeps family-level claims valid and forfeits per-model claims.
|
|
112
|
+
*/
|
|
113
|
+
allowWithinFamily?: boolean;
|
|
114
|
+
/**
|
|
115
|
+
* Accept a response that carried no model id. Default false — an
|
|
116
|
+
* unidentified response cannot support a per-model or per-family claim.
|
|
117
|
+
*/
|
|
118
|
+
allowUnreported?: boolean;
|
|
119
|
+
/** Prefixed to the thrown message, e.g. the judge or campaign cell name. */
|
|
120
|
+
context?: string;
|
|
121
|
+
}
|
|
122
|
+
/**
|
|
123
|
+
* The one place the accept/reject policy lives, so a caller that reports
|
|
124
|
+
* substitution (a preflight table, a run record) and a caller that throws on it
|
|
125
|
+
* can never drift apart. A cross-family substitution is never acceptable.
|
|
126
|
+
*/
|
|
127
|
+
declare function servedModelAcceptable(check: ServedModelCheck, opts?: AssertServedModelOptions): boolean;
|
|
128
|
+
/**
|
|
129
|
+
* Throw `ModelSubstitutionError` unless the served id is the requested model.
|
|
130
|
+
* Returns the check on success so callers can record the served id alongside
|
|
131
|
+
* the result.
|
|
132
|
+
*/
|
|
133
|
+
declare function assertServedModel(requested: string, served: string | null | undefined, opts?: AssertServedModelOptions): ServedModelCheck;
|
|
134
|
+
/**
|
|
135
|
+
* Batch form: check every pair and throw naming EVERY substitution, so one
|
|
136
|
+
* failure does not hide the rest. Returns all checks on success.
|
|
137
|
+
*/
|
|
138
|
+
declare function assertServedModels(pairs: ReadonlyArray<{
|
|
139
|
+
requested: string;
|
|
140
|
+
served: string | null | undefined;
|
|
141
|
+
}>, opts?: AssertServedModelOptions): ServedModelCheck[];
|
|
142
|
+
interface AssertCrossFamilyServedOptions extends AssertServedModelOptions {
|
|
143
|
+
/** Minimum distinct SERVED families required. Default 2. */
|
|
144
|
+
minFamilies?: number;
|
|
145
|
+
/** Count `unknown`-family served ids toward the total. Default false. */
|
|
146
|
+
allowUnknown?: boolean;
|
|
147
|
+
}
|
|
148
|
+
declare class ServedCrossFamilyError extends AgentEvalError {
|
|
149
|
+
readonly families: JudgeFamily[];
|
|
150
|
+
readonly checks: ReadonlyArray<ServedModelCheck>;
|
|
151
|
+
constructor(message: string, families: JudgeFamily[], checks: ReadonlyArray<ServedModelCheck>);
|
|
152
|
+
}
|
|
153
|
+
/**
|
|
154
|
+
* Family-diversity rule over the models that actually ANSWERED.
|
|
155
|
+
*
|
|
156
|
+
* `assertCrossFamily` (../judge-families) reads the requested ids and so
|
|
157
|
+
* cannot see a gateway that answers three "different" requests from one
|
|
158
|
+
* provider. This one asserts no substitution first, then counts families from
|
|
159
|
+
* the served ids — a panel that collapsed to one family under the hood fails
|
|
160
|
+
* here even though its request list looked diverse.
|
|
161
|
+
*/
|
|
162
|
+
declare function assertCrossFamilyServed(pairs: ReadonlyArray<{
|
|
163
|
+
requested: string;
|
|
164
|
+
served: string | null | undefined;
|
|
165
|
+
}>, opts?: AssertCrossFamilyServedOptions): JudgeFamily[];
|
|
166
|
+
//#endregion
|
|
11
167
|
//#region src/llm-judge.d.ts
|
|
12
168
|
/** A rubric dimension as a bare key or the full `{ key, description }` shape. A
|
|
13
169
|
* bare string uses the key as its own description. */
|
|
@@ -29,6 +185,27 @@ interface LlmJudgeOptions<TArtifact, TScenario extends Scenario = Scenario> {
|
|
|
29
185
|
/** Composite weights forwarded to `weightedComposite`: a partial map selects
|
|
30
186
|
* AND weights exactly the named dimensions. Omit for a uniform mean. */
|
|
31
187
|
weights?: Record<string, number>;
|
|
188
|
+
/**
|
|
189
|
+
* How to read a score out of the model's answer.
|
|
190
|
+
*
|
|
191
|
+
* `'sampled'` (default) reads the number the model emitted. Discrete grades
|
|
192
|
+
* tie often, and a tie carries no ranking signal.
|
|
193
|
+
*
|
|
194
|
+
* `'expectation'` asks the provider for the log probabilities of the score
|
|
195
|
+
* token and returns the expected value over the integer grades the model
|
|
196
|
+
* considered, so two answers that both sample `8` separate by how much mass
|
|
197
|
+
* sat on `7` and `9`. It requires `scale: 'ten'`: an integer grade is one
|
|
198
|
+
* token, and a `unit` float is not. `whenUnavailable` decides what happens
|
|
199
|
+
* when the provider returns no log probabilities, or the grade did not land
|
|
200
|
+
* in one token: `'fail'` throws, `'sampled'` reads the emitted number and
|
|
201
|
+
* records `scoringMethod: 'sampled'` on the score.
|
|
202
|
+
*/
|
|
203
|
+
scoring?: {
|
|
204
|
+
method: 'sampled';
|
|
205
|
+
} | {
|
|
206
|
+
method: 'expectation';
|
|
207
|
+
whenUnavailable: 'fail' | 'sampled';
|
|
208
|
+
};
|
|
32
209
|
/** Scale the model is prompted to score on, normalized into `[0,1]`:
|
|
33
210
|
* - `'unit'` (default): the model returns `[0,1]` directly.
|
|
34
211
|
* - `'ten'`: the model returns `[0,10]`; divided by 10 here.
|
|
@@ -257,8 +434,18 @@ interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
|
|
|
257
434
|
* rejects with the exact error thrown by that dispatch or judge.
|
|
258
435
|
* Default false preserves the normal behavior of returning failed cells and
|
|
259
436
|
* continuing the remaining schedule.
|
|
437
|
+
* With `cellRetry`, a retryable failure is not an error yet: this abort
|
|
438
|
+
* fires only when a cell's final attempt fails.
|
|
260
439
|
*/
|
|
261
440
|
abortOnCellError?: boolean;
|
|
441
|
+
/**
|
|
442
|
+
* Opt-in bounded in-run retry of a failed cell. Absent by default: a failed
|
|
443
|
+
* cell is final on its first attempt. A retried attempt re-runs the SAME
|
|
444
|
+
* slot — same `cellId`, same `seed`, same cost tags — so the schedule,
|
|
445
|
+
* manifest, and pairing are unchanged. An attempt that failed because the
|
|
446
|
+
* campaign was cancelled is never retried.
|
|
447
|
+
*/
|
|
448
|
+
cellRetry?: CampaignCellRetryPolicy;
|
|
262
449
|
/**
|
|
263
450
|
* Per-cell dispatch deadline in ms. A `dispatch` that neither resolves nor
|
|
264
451
|
* rejects within this window is a hang (a stalled model request, an
|
|
@@ -343,6 +530,25 @@ interface CampaignCellFailureReceipt<TArtifact = unknown> {
|
|
|
343
530
|
cell: CampaignCellResult<TArtifact>;
|
|
344
531
|
cost: CostLedgerSummary;
|
|
345
532
|
}
|
|
533
|
+
/**
|
|
534
|
+
* Bounded in-run retry of failed cells. Every attempt dispatches the same
|
|
535
|
+
* slot and charges the shared cost ledger, so the final cell's `costUsd`,
|
|
536
|
+
* `tokenUsage`, and `costCallIds` cover all attempts. Each retried attempt
|
|
537
|
+
* keeps its failure receipt at `<cell>/failure-receipt.attempt-<n>.json`; a
|
|
538
|
+
* final failed attempt keeps the usual `<cell>/failure-receipt.json`. The
|
|
539
|
+
* final cell records the retry count as `retryAttempts`. Artifacts and trace
|
|
540
|
+
* spans written by a later attempt replace those of the retried attempt; the
|
|
541
|
+
* per-attempt failure receipts are the durable evidence.
|
|
542
|
+
*/
|
|
543
|
+
interface CampaignCellRetryPolicy {
|
|
544
|
+
/** Total attempts per cell, including the first. A positive safe integer. */
|
|
545
|
+
attempts: number;
|
|
546
|
+
/** Decides whether a failed attempt is dispatched again. Receives the
|
|
547
|
+
* receipt's `failure` record. `transientDispatchFailure()` is the
|
|
548
|
+
* ready-made predicate for infrastructure hiccups (502/503/504, dropped
|
|
549
|
+
* streams, admission rejections). */
|
|
550
|
+
retryable: (failure: CampaignCellFailureReceipt['failure']) => boolean;
|
|
551
|
+
}
|
|
346
552
|
/**
|
|
347
553
|
* Core campaign orchestrator: fan scenarios through dispatch, score with judges, aggregate bootstrap CIs, and persist reproducible `CampaignResult` records.
|
|
348
554
|
*/
|
|
@@ -599,6 +805,17 @@ interface SearchPlannedEvent extends SearchLedgerEventBase {
|
|
|
599
805
|
kind: 'search-planned';
|
|
600
806
|
plan: SearchPlan;
|
|
601
807
|
}
|
|
808
|
+
/** Additional candidate slots and operations for a search whose length is not
|
|
809
|
+
* known when it starts. The plan stays the first event and the planned task
|
|
810
|
+
* denominator stays frozen: extending tasks would retroactively reopen
|
|
811
|
+
* candidates that already closed theirs. */
|
|
812
|
+
interface SearchPlanExtendedEvent extends SearchLedgerEventBase {
|
|
813
|
+
kind: 'search-plan-extended';
|
|
814
|
+
extension: {
|
|
815
|
+
candidateSlots: SearchCandidateSlot[];
|
|
816
|
+
operations: SearchPlannedOperation[];
|
|
817
|
+
};
|
|
818
|
+
}
|
|
602
819
|
interface SearchCandidateRegisteredEvent extends SearchLedgerEventBase {
|
|
603
820
|
kind: 'candidate-registered';
|
|
604
821
|
slotId: string;
|
|
@@ -674,7 +891,7 @@ interface SearchCompletedEvent extends SearchLedgerEventBase {
|
|
|
674
891
|
reason: SearchFailureReason;
|
|
675
892
|
};
|
|
676
893
|
}
|
|
677
|
-
type SearchLedgerEvent = SearchPlannedEvent | SearchCandidateRegisteredEvent | SearchCandidateSlotClosedEvent | SearchTaskAttemptedEvent | SearchOperationRecordedEvent | SearchCandidateDecidedEvent | SearchCompletedEvent;
|
|
894
|
+
type SearchLedgerEvent = SearchPlannedEvent | SearchPlanExtendedEvent | SearchCandidateRegisteredEvent | SearchCandidateSlotClosedEvent | SearchTaskAttemptedEvent | SearchOperationRecordedEvent | SearchCandidateDecidedEvent | SearchCompletedEvent;
|
|
678
895
|
interface SearchLedgerEntry {
|
|
679
896
|
schema: typeof SEARCH_LEDGER_SCHEMA;
|
|
680
897
|
campaignId: string;
|
|
@@ -736,6 +953,9 @@ interface SearchLedgerAudit {
|
|
|
736
953
|
interface SearchLedgerReplay {
|
|
737
954
|
entries: SearchLedgerEntry[];
|
|
738
955
|
plan: SearchPlannedEvent | null;
|
|
956
|
+
/** Appended plan extensions, in ledger order. The effective plan is the
|
|
957
|
+
* first plan event merged with these; `audit.expected` counts the merge. */
|
|
958
|
+
planExtensions: SearchPlanExtendedEvent[];
|
|
739
959
|
candidates: SearchCandidateRegisteredEvent[];
|
|
740
960
|
closedCandidateSlots: SearchCandidateSlotClosedEvent[];
|
|
741
961
|
attempts: SearchTaskAttemptedEvent[];
|
|
@@ -1251,6 +1471,183 @@ declare function scoreRedTeamOutput(output: string, toolCalls: string[], rtCase:
|
|
|
1251
1471
|
/** Aggregate red-team findings into per-category pass rates. */
|
|
1252
1472
|
declare function redTeamReport(findings: RedTeamFinding[]): RedTeamReport;
|
|
1253
1473
|
//#endregion
|
|
1474
|
+
//#region src/campaign/search-ledger-recording.d.ts
|
|
1475
|
+
/** How a search operation executed. The shape the ledger event records. */
|
|
1476
|
+
type SearchExecutionIdentity = SearchOperationRecordedEvent['execution'];
|
|
1477
|
+
/** Immutable identities the ledger requires and a campaign cannot infer. */
|
|
1478
|
+
interface SearchRunIdentity {
|
|
1479
|
+
/** The agent implementation under optimization. */
|
|
1480
|
+
agent: SearchSourceRef;
|
|
1481
|
+
/** The candidate generator: a model call or deterministic code. */
|
|
1482
|
+
proposer: SearchExecutionIdentity;
|
|
1483
|
+
/** The code that plans the search and selects its winner. */
|
|
1484
|
+
search: SearchSourceRef;
|
|
1485
|
+
/** Model the agent runs. Used only for a cell that reported none. */
|
|
1486
|
+
model: SearchModelIdentity;
|
|
1487
|
+
}
|
|
1488
|
+
interface SearchLedgerBinding {
|
|
1489
|
+
ledger: SearchLedger;
|
|
1490
|
+
identity: SearchRunIdentity;
|
|
1491
|
+
}
|
|
1492
|
+
/** One proposed candidate, before it is measured. */
|
|
1493
|
+
interface ProposedSearchCandidate {
|
|
1494
|
+
surface: MutableSurface;
|
|
1495
|
+
surfaceHash: string;
|
|
1496
|
+
label?: string;
|
|
1497
|
+
}
|
|
1498
|
+
/** One measured candidate, after its campaign scored. */
|
|
1499
|
+
interface MeasuredSearchCandidate<TArtifact> {
|
|
1500
|
+
surface: MutableSurface;
|
|
1501
|
+
surfaceHash: string;
|
|
1502
|
+
cells: ReadonlyArray<CampaignCellResult<TArtifact>>;
|
|
1503
|
+
runDir: string;
|
|
1504
|
+
/** False when the candidate missed a designed cell. */
|
|
1505
|
+
coverageComplete: boolean;
|
|
1506
|
+
}
|
|
1507
|
+
interface SearchRecorderOptions<TScenario extends Scenario> {
|
|
1508
|
+
binding: SearchLedgerBinding;
|
|
1509
|
+
storage: CampaignStorage;
|
|
1510
|
+
runDir: string;
|
|
1511
|
+
scenarios: ReadonlyArray<TScenario>;
|
|
1512
|
+
reps: number;
|
|
1513
|
+
maxGenerations: number;
|
|
1514
|
+
populationSize: number;
|
|
1515
|
+
/** Identity of the exact campaign design; the task benchmark pin. */
|
|
1516
|
+
splitDigest: `sha256:${string}`;
|
|
1517
|
+
/** Proposer label recorded on every candidate lineage. */
|
|
1518
|
+
proposerLabel: string;
|
|
1519
|
+
costLedger: CostLedgerHandle;
|
|
1520
|
+
}
|
|
1521
|
+
/**
|
|
1522
|
+
* Recorder for one `runOptimization` run: `open()`, then `recordGeneration()`
|
|
1523
|
+
* and `recordResults()` per generation, then `finish()`.
|
|
1524
|
+
*
|
|
1525
|
+
* Every event id is derived from the run, and an id already durable is not
|
|
1526
|
+
* appended again, so a resumed run continues one ledger instead of conflicting
|
|
1527
|
+
* with its own history.
|
|
1528
|
+
*/
|
|
1529
|
+
declare class SearchRecorder<TScenario extends Scenario, TArtifact> {
|
|
1530
|
+
private readonly opts;
|
|
1531
|
+
private readonly tasks;
|
|
1532
|
+
private readonly registered;
|
|
1533
|
+
private readonly order;
|
|
1534
|
+
private readonly coverage;
|
|
1535
|
+
private readonly openSlots;
|
|
1536
|
+
private readonly durableEventIds;
|
|
1537
|
+
private lastStampMs;
|
|
1538
|
+
private proposalReceiptCount;
|
|
1539
|
+
private constructor();
|
|
1540
|
+
/** Open the recorder and append the plan. An existing ledger for the same
|
|
1541
|
+
* run is re-read first, so a resumed run keeps one plan and one lineage. */
|
|
1542
|
+
static open<TScenario extends Scenario, TArtifact>(opts: SearchRecorderOptions<TScenario>): Promise<SearchRecorder<TScenario, TArtifact>>;
|
|
1543
|
+
/**
|
|
1544
|
+
* Record one generation's candidate-generation call and the candidates it
|
|
1545
|
+
* produced. A proposal larger than the planned population extends the plan
|
|
1546
|
+
* with the extra slots; a proposal that fills fewer closes the rest.
|
|
1547
|
+
*/
|
|
1548
|
+
recordGeneration(input: {
|
|
1549
|
+
generation: number;
|
|
1550
|
+
parentSurfaceHash: string;
|
|
1551
|
+
candidates: ReadonlyArray<ProposedSearchCandidate>;
|
|
1552
|
+
}): Promise<void>;
|
|
1553
|
+
/** Append one task attempt per designed cell of each candidate campaign. */
|
|
1554
|
+
recordResults(candidates: ReadonlyArray<MeasuredSearchCandidate<TArtifact>>): Promise<void>;
|
|
1555
|
+
/**
|
|
1556
|
+
* Close the search: unreached generations, the selection operation, one
|
|
1557
|
+
* decision per candidate, then the terminal event.
|
|
1558
|
+
*
|
|
1559
|
+
* The terminal event is appended only when canonical replay accounts for the
|
|
1560
|
+
* whole planned denominator. An interrupted or partly unscored search stays
|
|
1561
|
+
* `in-progress` and its receipt reports the exact gap, instead of claiming a
|
|
1562
|
+
* closed search.
|
|
1563
|
+
*/
|
|
1564
|
+
finish(input: {
|
|
1565
|
+
winnerSurfaceHash: string;
|
|
1566
|
+
generationsRun: number;
|
|
1567
|
+
runId: string;
|
|
1568
|
+
}): Promise<SearchHistoryReceipt>;
|
|
1569
|
+
/** Bounded receipt over the exact ledger bytes this run produced. */
|
|
1570
|
+
receipt(runId: string): Promise<SearchHistoryReceipt>;
|
|
1571
|
+
/** Read an existing ledger for this run so a resume continues it. */
|
|
1572
|
+
private hydrate;
|
|
1573
|
+
private plan;
|
|
1574
|
+
private registeredSlot;
|
|
1575
|
+
private remember;
|
|
1576
|
+
private recordGenerationOperation;
|
|
1577
|
+
private closeSlot;
|
|
1578
|
+
/** Spend booked to candidate generation since the previous generation. */
|
|
1579
|
+
private proposalAccounting;
|
|
1580
|
+
private cellModel;
|
|
1581
|
+
private proposalArtifact;
|
|
1582
|
+
/** Write one canonical evidence document and return its content address. */
|
|
1583
|
+
private writeArtifact;
|
|
1584
|
+
private append;
|
|
1585
|
+
/** Non-decreasing ISO stamps; the ledger refuses an event that moves back. */
|
|
1586
|
+
private stamp;
|
|
1587
|
+
}
|
|
1588
|
+
/**
|
|
1589
|
+
* Record an optimizer's own candidate graph into the same ledger.
|
|
1590
|
+
*
|
|
1591
|
+
* A complete optimization method searches inside its own process and reports
|
|
1592
|
+
* one artifact when it finishes: the candidate population, with each
|
|
1593
|
+
* candidate's parents and its score per selection scenario. This turns that
|
|
1594
|
+
* artifact into the canonical event stream, so a first-party method returns
|
|
1595
|
+
* the same `SearchHistoryReceipt` the in-process loop returns, and
|
|
1596
|
+
* `compareOptimizationMethods({ searchHistoryPolicy: 'require-complete' })`
|
|
1597
|
+
* accepts it.
|
|
1598
|
+
*
|
|
1599
|
+
* A candidate the optimizer left unscored on a planned scenario leaves the
|
|
1600
|
+
* planned denominator open, so the receipt reports the gap instead of closing
|
|
1601
|
+
* the search.
|
|
1602
|
+
*/
|
|
1603
|
+
declare function recordCandidatePopulationSearch<TScenario extends Scenario>(input: {
|
|
1604
|
+
ledger: SearchLedger;
|
|
1605
|
+
storage: CampaignStorage;
|
|
1606
|
+
runDir: string;
|
|
1607
|
+
identity: SearchRunIdentity;
|
|
1608
|
+
population: GepaCandidatePopulationArtifact;
|
|
1609
|
+
/** Scenarios the optimizer selected on. Must cover the population's ids. */
|
|
1610
|
+
scenarios: ReadonlyArray<TScenario>;
|
|
1611
|
+
/** Spend the optimizer booked to its own candidate generation. */
|
|
1612
|
+
generationAccounting: SearchAttemptAccounting;
|
|
1613
|
+
producerId: string;
|
|
1614
|
+
runId: string;
|
|
1615
|
+
}): Promise<SearchHistoryReceipt>;
|
|
1616
|
+
//#endregion
|
|
1617
|
+
//#region src/campaign/parent-selection.d.ts
|
|
1618
|
+
/** Search state supplied to one parent-selection call. */
|
|
1619
|
+
interface ParentSelectionContext {
|
|
1620
|
+
/** Non-dominated scored surfaces across the whole run so far, including the
|
|
1621
|
+
* baseline (`generation: -1`). Never empty. */
|
|
1622
|
+
readonly frontier: ReadonlyArray<ParetoParent>;
|
|
1623
|
+
/** Measured result of the global incumbent, the promotion bar. Under the
|
|
1624
|
+
* default `selectionRankKey` the incumbent is always on `frontier`. */
|
|
1625
|
+
readonly incumbent: ScoredSurfaceOutcome;
|
|
1626
|
+
/** Every completed generation so far. */
|
|
1627
|
+
readonly history: ReadonlyArray<GenerationRecord>;
|
|
1628
|
+
/** Index of the generation about to propose. */
|
|
1629
|
+
readonly generation: number;
|
|
1630
|
+
}
|
|
1631
|
+
/** Chooses the surface the next generation mutates. Returns one frontier
|
|
1632
|
+
* parent; `runOptimization` refuses a parent it has not measured to
|
|
1633
|
+
* completion or whose surface does not match its `surfaceHash`. */
|
|
1634
|
+
type ParentSelector = (ctx: ParentSelectionContext) => ParetoParent;
|
|
1635
|
+
interface CrowdedFrontierParentOptions {
|
|
1636
|
+
/** Integer seed for the per-generation draw. The same seed, frontier, and
|
|
1637
|
+
* generation index select the same parent. */
|
|
1638
|
+
seed: number;
|
|
1639
|
+
}
|
|
1640
|
+
/**
|
|
1641
|
+
* NSGA-II crowded tournament selection over the frontier. Each generation
|
|
1642
|
+
* draws two distinct frontier members with a PRNG seeded from `seed` and the
|
|
1643
|
+
* generation index, and keeps the one with the larger crowding distance (more
|
|
1644
|
+
* isolated on the frontier). Boundary parents carry infinite distance, so a
|
|
1645
|
+
* boundary parent always beats an interior one. A tie on distance falls back
|
|
1646
|
+
* to the higher mean composite, then to the smaller surface hash. A frontier
|
|
1647
|
+
* of one member returns that member.
|
|
1648
|
+
*/
|
|
1649
|
+
declare function crowdedFrontierParent(options: CrowdedFrontierParentOptions): ParentSelector;
|
|
1650
|
+
//#endregion
|
|
1254
1651
|
//#region src/campaign/presets/run-eval.d.ts
|
|
1255
1652
|
interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
|
|
1256
1653
|
runDir: string;
|
|
@@ -1334,6 +1731,31 @@ interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> exte
|
|
|
1334
1731
|
* vectors are untouched, so proposer diversity and reporting are unaffected.
|
|
1335
1732
|
*/
|
|
1336
1733
|
selectionRankKey?: (campaign: CampaignResult<TArtifact, TScenario>) => number[];
|
|
1734
|
+
/**
|
|
1735
|
+
* Optional policy for which scored surface the next generation MUTATES.
|
|
1736
|
+
* Absent, every generation mutates the global incumbent, so the recorded
|
|
1737
|
+
* `parentSurfaceHash` lineage is a chain. Present, the selector receives the
|
|
1738
|
+
* Pareto frontier so far, the measured incumbent, the generation history,
|
|
1739
|
+
* and the generation index, and returns one frontier parent; the loop hands
|
|
1740
|
+
* that parent to `propose()` as `currentSurface` + `parentOutcome` and
|
|
1741
|
+
* records it as every candidate's `parentSurfaceHash`. Promotion is
|
|
1742
|
+
* unchanged: a candidate still has to beat the incumbent. The loop refuses
|
|
1743
|
+
* a parent it has not measured to completion. `crowdedFrontierParent` is
|
|
1744
|
+
* the provided seeded policy.
|
|
1745
|
+
*/
|
|
1746
|
+
selectParent?: ParentSelector;
|
|
1747
|
+
/**
|
|
1748
|
+
* Record this search into a durable `SearchLedger`. The loop emits the plan,
|
|
1749
|
+
* each candidate-generation operation, each candidate registration with its
|
|
1750
|
+
* measured parent, one task attempt per designed cell, one decision per
|
|
1751
|
+
* candidate, and the terminal event, then returns a bounded
|
|
1752
|
+
* `searchHistory` receipt over the exact ledger bytes.
|
|
1753
|
+
*
|
|
1754
|
+
* `identity` declares what the ledger requires and a campaign cannot infer:
|
|
1755
|
+
* immutable revisions for the agent, proposer, and search implementations,
|
|
1756
|
+
* and the model the agent runs when a cell reports none.
|
|
1757
|
+
*/
|
|
1758
|
+
searchLedger?: SearchLedgerBinding;
|
|
1337
1759
|
}
|
|
1338
1760
|
type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
|
|
1339
1761
|
interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
|
|
@@ -1360,6 +1782,10 @@ interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
|
|
|
1360
1782
|
baselineCampaign: CampaignResult<TArtifact, TScenario>;
|
|
1361
1783
|
/** Run-wide spend, including agents, proposers, analysts, and judges. */
|
|
1362
1784
|
cost: CostLedgerSummary;
|
|
1785
|
+
/** Bounded proof envelope over the canonical search ledger. Present only
|
|
1786
|
+
* when `searchLedger` was supplied. `complete` is false when the search was
|
|
1787
|
+
* interrupted or a candidate left a designed cell unscored. */
|
|
1788
|
+
searchHistory?: SearchHistoryReceipt;
|
|
1363
1789
|
/** The GEPA Pareto frontier across every scored surface (baseline + all
|
|
1364
1790
|
* generations) by per-scenario objective vector — the non-dominated set.
|
|
1365
1791
|
* Each generation's `propose()` received the frontier-so-far as
|
|
@@ -1368,7 +1794,7 @@ interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
|
|
|
1368
1794
|
paretoFrontier: ParetoParent[];
|
|
1369
1795
|
}
|
|
1370
1796
|
/**
|
|
1371
|
-
* Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations.
|
|
1797
|
+
* Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations. The parent each generation mutates is the incumbent unless `selectParent` draws it from the frontier.
|
|
1372
1798
|
*/
|
|
1373
1799
|
declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
|
|
1374
1800
|
//#endregion
|
|
@@ -1602,7 +2028,8 @@ declare function buildLoopProvenanceRecord<TArtifact, TScenario extends Scenario
|
|
|
1602
2028
|
declare function campaignMeasurementDigest<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): `sha256:${string}`;
|
|
1603
2029
|
/** Recompute and validate the self-addressed durable record. */
|
|
1604
2030
|
declare function verifyLoopProvenanceRecord(record: LoopProvenanceRecord): LoopProvenanceRecord;
|
|
1605
|
-
/**
|
|
2031
|
+
/** SHA-256 over the RFC 8785 canonical JSON of `value`. Throws
|
|
2032
|
+
* `LedgerCanonicalizationError` for a value with no canonical form. */
|
|
1606
2033
|
declare function canonicalDigest(value: unknown): `sha256:${string}`;
|
|
1607
2034
|
/**
|
|
1608
2035
|
* Build the loop's OTLP-ingestable spans from a provenance record. One root
|
|
@@ -1650,5 +2077,33 @@ interface EmitLoopProvenanceArgs<TArtifact, TScenario extends Scenario> extends
|
|
|
1650
2077
|
*/
|
|
1651
2078
|
declare function emitLoopProvenance<TArtifact, TScenario extends Scenario>(args: EmitLoopProvenanceArgs<TArtifact, TScenario>): Promise<EmitLoopProvenanceResult>;
|
|
1652
2079
|
//#endregion
|
|
1653
|
-
|
|
1654
|
-
|
|
2080
|
+
//#region src/campaign/transient-failure.d.ts
|
|
2081
|
+
interface TransientFailureOptions {
|
|
2082
|
+
/**
|
|
2083
|
+
* Treat full-duration timeouts ("timeout after 180000ms") as transient.
|
|
2084
|
+
* Enable on saturated shared infrastructure where queue starvation eats
|
|
2085
|
+
* the clock; leave off when the agent had the resources and simply failed.
|
|
2086
|
+
* Default false.
|
|
2087
|
+
*/
|
|
2088
|
+
readonly retryFullDurationTimeouts?: boolean;
|
|
2089
|
+
/** Additional caller-specific transient patterns. */
|
|
2090
|
+
readonly extraPatterns?: readonly RegExp[];
|
|
2091
|
+
}
|
|
2092
|
+
/**
|
|
2093
|
+
* True when the error text describes an infrastructure hiccup that should be
|
|
2094
|
+
* retried rather than scored. Empty/undefined input is not transient.
|
|
2095
|
+
*/
|
|
2096
|
+
declare function isTransientTransportFailure(message: string | null | undefined, opts?: TransientFailureOptions): boolean;
|
|
2097
|
+
/**
|
|
2098
|
+
* Ready-made `cellRetry.retryable` predicate: true for a dispatch-stage
|
|
2099
|
+
* failure whose error message `isTransientTransportFailure` classifies as an
|
|
2100
|
+
* infrastructure hiccup. A judge-stage failure is never retried here — the
|
|
2101
|
+
* dispatch already produced an artifact, so re-dispatching would score a
|
|
2102
|
+
* different sample. A per-cell dispatch deadline ("dispatch exceeded <N>ms")
|
|
2103
|
+
* is not transient by default; opt in via `extraPatterns` or
|
|
2104
|
+
* `retryFullDurationTimeouts` when queue starvation eats the clock.
|
|
2105
|
+
*/
|
|
2106
|
+
declare function transientDispatchFailure(opts?: TransientFailureOptions): (failure: CampaignCellFailureReceipt['failure']) => boolean;
|
|
2107
|
+
//#endregion
|
|
2108
|
+
export { CanaryKind as $, ServedModelVerdict as $n, SearchOperationKind as $t, runEval as A, CampaignCellRetryPolicy as An, verifySearchHistoryReceipt as At, SearchRecorderOptions as B, inMemoryCampaignStorage as Bn, SearchCandidateSlotClosedEvent as Bt, RunImprovementLoopResult as C, ExternalOptimizerSubmittedCandidate as Cn, SearchHistoryPolicy as Ct, RunOptimizationResult as D, readCachedCell as Dn, assertSearchHistoryMatchesReplay as Dt, RunOptimizationOptions as E, CacheRead as En, assertCompleteSearchHistory as Et, MeasuredSearchCandidate as F, PlanCampaignRunOptions as Fn, SearchAttemptAccounting as Ft, RedTeamCategory as G, LlmJudgeOptions as Gn, SearchLedger as Gt, recordCandidatePopulationSearch as H, OpenAutoPrResult as Hn, SearchCompletedEvent as Ht, ProposedSearchCandidate as I, planCampaignRun as In, SearchCandidateDecidedEvent as It, redTeamDataset as J, AssertServedModelOptions as Jn, SearchLedgerEvent as Jt, RedTeamFinding as K, llmJudge as Kn, SearchLedgerAppendResult as Kt, SearchExecutionIdentity as L, CampaignStorage as Ln, SearchCandidateLineage as Lt, ParentSelectionContext as M, runCampaign as Mn, OpenSearchLedgerOptions as Mt, ParentSelector as N, CampaignRunPlan as Nn, SearchAccountingAudit as Nt, runOptimization as O, cellCachePath as On, createSearchHistoryReceipt as Ot, crowdedFrontierParent as P, CampaignRunPlanCell as Pn, SearchArtifactRef as Pt, CanaryEvaluation as Q, ServedModelPolicy as Qn, SearchModelIdentity as Qt, SearchLedgerBinding as R, createRunCostLedger as Rn, SearchCandidateRegisteredEvent as Rt, RunImprovementLoopOptions as S, ExternalOptimizerObservationSummary as Sn, SearchHistoryCoverageRow as St, PremeasuredOptimizationBaseline as T, CacheIssueReason as Tn, SearchHistoryRequiredError as Tt, DEFAULT_RED_TEAM_CORPUS as U, openAutoPr as Un, SearchCostAccounting as Ut, SearchRunIdentity as V, OpenAutoPrOptions as Vn, SearchCandidateSurface as Vt, RedTeamCase as W, LlmJudgeDimension as Wn, SearchFailureReason as Wt, scoreRedTeamOutput as X, ServedCrossFamilyError as Xn, SearchLedgerReplay as Xt, redTeamReport as Y, ModelSubstitutionError as Yn, SearchLedgerHash as Yt, CanaryAlert as Z, ServedModelCheck as Zn, SearchLedgerTrustedHeadMode as Zt, loopProvenanceArgsFromResult as _, GepaCandidatePopulationSummary as _n, costFromLedgerSummary as _t, EmitLoopProvenanceArgs as a, SearchPlannedTask as an, AssertCrossFamilyOptions as ar, OptimizationMethod as at, provenanceSpansPath as b, ExternalOptimizerExecutionSummary as bn, SearchHistoryAuditSummary as bt, LoopProvenanceBackend as c, SearchSurfaceEvidence as cn, assertCrossFamily as cr, OptimizationMethodPairwise as ct, LoopProvenanceOptimizationMethod as d, SearchTaskOutcome as dn, OptimizationMethodRunOptions as dt, SearchOperationRecordedEvent as en, assertCrossFamilyServed as er, CanaryOptions as et, LoopProvenanceRecord as f, SearchTokenAccounting as fn, OptimizationMethodScore as ft, emitLoopProvenance as g, GepaCandidatePopulationCandidate as gn, compareOptimizationMethods as gt, canonicalDigest as h, GepaCandidatePopulationArtifact as hn, combineComparisonCosts as ht, BuildLoopProvenanceArgs as i, SearchPlannedOperation as in, servedModelAcceptable as ir, ComparisonCost as it, CrowdedFrontierParentOptions as j, RunCampaignOptions as jn, FileSearchLedger as jt, RunEvalOptions as k, CampaignCellFailureReceipt as kn, searchHistoryCoverageRow as kt, LoopProvenanceCandidate as l, SearchSurfaceKind as ln, judgeFamily as lr, OptimizationMethodProvenance as lt, campaignMeasurementDigest as m, validateSearchLedgerEvent as mn, OptimizationTokenUsage as mt, isTransientTransportFailure as n, SearchPlanExtendedEvent as nn, assertServedModels as nr, runCanaries as nt, EmitLoopProvenanceResult as o, SearchSourceRef as on, CrossFamilyError as or, OptimizationMethodComparison as ot, buildLoopProvenanceRecord as p, openSearchLedger as pn, OptimizationPackageSource as pt, RedTeamReport as q, AssertCrossFamilyServedOptions as qn, SearchLedgerEntry as qt, transientDispatchFailure as r, SearchPlannedEvent as rn, checkServedModel as rr, CompareOptimizationMethodsOptions as rt, LoopProvenanceArgsFromResult as s, SearchSurfaceEffect as sn, JudgeFamily as sr, OptimizationMethodInput as st, TransientFailureOptions as t, SearchPlan as tn, assertServedModel as tr, CanaryReport as tt, LoopProvenanceEvidence as u, SearchTaskAttemptedEvent as un, OptimizationMethodResult as ut, loopProvenanceSpans as v, GepaCandidateSelectionScore as vn, optimizationTokenUsageFromSummary as vt, runImprovementLoop as w, readExternalOptimizerObservationArtifact as wn, SearchHistoryReceipt as wt, verifyLoopProvenanceRecord as x, ExternalOptimizerObservationArtifact as xn, SearchHistoryCoverage as xt, provenanceRecordPath as y, readGepaCandidatePopulationArtifact as yn, CreateSearchHistoryReceiptInput as yt, SearchRecorder as z, fsCampaignStorage as zn, SearchCandidateSlot as zt };
|
|
2109
|
+
//# sourceMappingURL=transient-failure-DKF5Mofa.d.ts.map
|