@tangle-network/agent-eval 0.163.2 → 0.171.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +127 -0
- package/README.md +2 -0
- package/dist/adapters/http.d.ts +108 -0
- package/dist/adapters/http.d.ts.map +1 -0
- package/dist/adapters/http.js +208 -0
- package/dist/adapters/http.js.map +1 -0
- package/dist/analyst/index.d.ts +40 -70
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +18 -311
- package/dist/analyst/index.js.map +1 -1
- package/dist/{backend-integrity-DxuQCu_A.d.ts → backend-integrity-e79K3UPD.d.ts} +3 -3
- package/dist/{backend-integrity-DxuQCu_A.d.ts.map → backend-integrity-e79K3UPD.d.ts.map} +1 -1
- package/dist/{benchmark-BhT16ep9.js → benchmark-C4wk_Sjr.js} +10 -3
- package/dist/benchmark-C4wk_Sjr.js.map +1 -0
- package/dist/{benchmark-command-CF-4GEWZ.js → benchmark-command-D8k3Gf0J.js} +236 -251
- package/dist/benchmark-command-D8k3Gf0J.js.map +1 -0
- package/dist/{benchmark-CGPp-kDC.d.ts → benchmark-h-h4bfqj.d.ts} +3 -3
- package/dist/{benchmark-CGPp-kDC.d.ts.map → benchmark-h-h4bfqj.d.ts.map} +1 -1
- package/dist/benchmarks/index.d.ts +5 -5
- package/dist/benchmarks/index.js +3 -3
- package/dist/builder-eval/index.d.ts +3 -3
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +8 -8
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-DQZmc2Dq.js → campaign-B72njjHj.js} +15 -14
- package/dist/campaign-B72njjHj.js.map +1 -0
- package/dist/{canonical-IL-Bu-14.js → canonical-DPyQ_rpt.js} +22 -2
- package/dist/{canonical-IL-Bu-14.js.map → canonical-DPyQ_rpt.js.map} +1 -1
- package/dist/{chat-client-DlMlAeYI.js → chat-client-DEtybj5i.js} +5 -5
- package/dist/{chat-client-DlMlAeYI.js.map → chat-client-DEtybj5i.js.map} +1 -1
- package/dist/cli.js +2 -2
- package/dist/{client-CX7KqIdB.js → client-BvwNkIRN.js} +2 -2
- package/dist/{client-CX7KqIdB.js.map → client-BvwNkIRN.js.map} +1 -1
- package/dist/{client-L9VVPkim.d.ts → client-CDtcZ3p9.d.ts} +4 -4
- package/dist/{client-L9VVPkim.d.ts.map → client-CDtcZ3p9.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +12 -703
- package/dist/contract/index.js +11 -11
- package/dist/{counterfactual-BaFUWK3H.d.ts → counterfactual-Bee5_BIn.d.ts} +4 -4
- package/dist/{counterfactual-BaFUWK3H.d.ts.map → counterfactual-Bee5_BIn.d.ts.map} +1 -1
- package/dist/{default-registry-G9CKMNkc.d.ts → default-registry-B0s2zU-s.d.ts} +6 -6
- package/dist/{default-registry-G9CKMNkc.d.ts.map → default-registry-B0s2zU-s.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Dx1JnPEa.d.ts → define-agent-eval-Cjy2yhqP.d.ts} +7 -7
- package/dist/{define-agent-eval-Dx1JnPEa.d.ts.map → define-agent-eval-Cjy2yhqP.d.ts.map} +1 -1
- package/dist/{define-agent-eval-D08pWIJb.js → define-agent-eval-Dy8QgxAI.js} +20 -8
- package/dist/{define-agent-eval-D08pWIJb.js.map → define-agent-eval-Dy8QgxAI.js.map} +1 -1
- package/dist/{dspy-rlm-engine-DhA9qKIm.js → dspy-rlm-engine-CS3qcCEk.js} +3 -10
- package/dist/dspy-rlm-engine-CS3qcCEk.js.map +1 -0
- package/dist/{emitter-D_jYSGRd.d.ts → emitter-Bvnu0VzL.d.ts} +3 -3
- package/dist/{emitter-D_jYSGRd.d.ts.map → emitter-Bvnu0VzL.d.ts.map} +1 -1
- package/dist/{engine-Cu5qD5Fc.d.ts → engine-CAmTUk52.d.ts} +7 -7
- package/dist/{engine-Cu5qD5Fc.d.ts.map → engine-CAmTUk52.d.ts.map} +1 -1
- package/dist/{eval-campaign-BfohKmzx.js → eval-campaign-JDTeE6Pl.js} +4 -4
- package/dist/{eval-campaign-BfohKmzx.js.map → eval-campaign-JDTeE6Pl.js.map} +1 -1
- package/dist/{exact-types-qnexxJ1Z.d.ts → exact-types-BEecmnWm.d.ts} +2 -2
- package/dist/{exact-types-qnexxJ1Z.d.ts.map → exact-types-BEecmnWm.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +5 -5
- package/dist/experiment/index.js +4 -4
- package/dist/{experiment-tracker-DCO6Cz4s.d.ts → experiment-tracker-Dm8yQMqb.d.ts} +2 -2
- package/dist/{experiment-tracker-DCO6Cz4s.d.ts.map → experiment-tracker-Dm8yQMqb.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-BFmh36vW.js → external-optimizer-process-CQxylYeG.js} +4 -11
- package/dist/external-optimizer-process-CQxylYeG.js.map +1 -0
- package/dist/{external-optimizer-subprocess-CqLMW3nh.js → external-optimizer-subprocess-Cex8Da2i.js} +25 -11
- package/dist/external-optimizer-subprocess-Cex8Da2i.js.map +1 -0
- package/dist/{failure-cluster-CXL8NbEw.d.ts → failure-cluster-6YSvsKlp.d.ts} +3 -3
- package/dist/{failure-cluster-CXL8NbEw.d.ts.map → failure-cluster-6YSvsKlp.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-B3ZHaHV_.d.ts → feedback-trajectory-DIqpCyF0.d.ts} +6 -6
- package/dist/{feedback-trajectory-B3ZHaHV_.d.ts.map → feedback-trajectory-DIqpCyF0.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/{skillopt-optimization-method-x7TTF23P.d.ts → heldout-gate-Dh2b62w8.d.ts} +111 -111
- package/dist/heldout-gate-Dh2b62w8.d.ts.map +1 -0
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.js +1 -1
- package/dist/{index-D-V8gCs_.d.ts → index-8VIogTyS.d.ts} +30 -22
- package/dist/index-8VIogTyS.d.ts.map +1 -0
- package/dist/{index-D-IiQIBB.d.ts → index-DMoxLG8P.d.ts} +3 -3
- package/dist/{index-D-IiQIBB.d.ts.map → index-DMoxLG8P.d.ts.map} +1 -1
- package/dist/{index-D_P7Ye43.d.ts → index-DNgf5gyG.d.ts} +2 -2
- package/dist/{index-D_P7Ye43.d.ts.map → index-DNgf5gyG.d.ts.map} +1 -1
- package/dist/{index-CGtH1piv.d.ts → index-DT73JraI.d.ts} +7 -38
- package/dist/index-DT73JraI.d.ts.map +1 -0
- package/dist/index-fNXZMCzX.d.ts +704 -0
- package/dist/index-fNXZMCzX.d.ts.map +1 -0
- package/dist/index.d.ts +197 -38
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +736 -28
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DRe8LB6d.d.ts → insight-report-08F022xN.d.ts} +4 -4
- package/dist/{insight-report-DRe8LB6d.d.ts.map → insight-report-08F022xN.d.ts.map} +1 -1
- package/dist/{integrity-DUNX9Fao.d.ts → integrity-B_EDELom.d.ts} +2 -2
- package/dist/{integrity-DUNX9Fao.d.ts.map → integrity-B_EDELom.d.ts.map} +1 -1
- package/dist/internal-BMFSR8Ns.js.map +1 -1
- package/dist/{kind-factory-DY8FdoXf.js → kind-factory-DMeEoMQZ.js} +3 -10
- package/dist/kind-factory-DMeEoMQZ.js.map +1 -0
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-BOzlRygb.js → ledger-core-PIfjCbKn.js} +2 -2
- package/dist/{ledger-core-BOzlRygb.js.map → ledger-core-PIfjCbKn.js.map} +1 -1
- package/dist/{llm-judge-Du7WQPh7.js → llm-judge-BtJ2Sfk_.js} +101 -20
- package/dist/llm-judge-BtJ2Sfk_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/{matrix-eXKRMHnL.d.ts → matrix-DiHmUobV.d.ts} +3 -3
- package/dist/{matrix-eXKRMHnL.d.ts.map → matrix-DiHmUobV.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +1 -1
- package/dist/{mint-DfODW1KW.js → mint-DjfDUMHr.js} +2 -2
- package/dist/{mint-DfODW1KW.js.map → mint-DjfDUMHr.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +4 -4
- package/dist/pipelines/index.d.ts +5 -5
- package/dist/pipelines/index.js +3 -3
- package/dist/{pareto-BqNW3LJR.d.ts → power-preflight-Ptse_Kq7.d.ts} +43 -43
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +1 -0
- package/dist/{pre-registration-KN9jkh58.js → pre-registration-D94b7Of5.js} +2 -2
- package/dist/{pre-registration-KN9jkh58.js.map → pre-registration-D94b7Of5.js.map} +1 -1
- package/dist/{pre-registration-CzFCcwYk.d.ts → pre-registration-DHz6P_6f.d.ts} +2 -2
- package/dist/{pre-registration-CzFCcwYk.d.ts.map → pre-registration-DHz6P_6f.d.ts.map} +1 -1
- package/dist/{produced-state-Be0BK3RN.js → produced-state-Cm6DU_Ao.js} +4 -4
- package/dist/{produced-state-Be0BK3RN.js.map → produced-state-Cm6DU_Ao.js.map} +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DtnOIZvk.d.ts → promotion-policy-WSXtBgBb.d.ts} +3 -3
- package/dist/{promotion-policy-DtnOIZvk.d.ts.map → promotion-policy-WSXtBgBb.d.ts.map} +1 -1
- package/dist/{run-score-lDzV0X8j.js → proposal-findings-bko3GGy-.js} +2 -31
- package/dist/proposal-findings-bko3GGy-.js.map +1 -0
- package/dist/{transient-failure-DKF5Mofa.d.ts → provenance-CafMdZKM.d.ts} +895 -891
- package/dist/provenance-CafMdZKM.d.ts.map +1 -0
- package/dist/{query-_5g6re3_.js → query-BPGMVlbM.js} +3 -3
- package/dist/{query-_5g6re3_.js.map → query-BPGMVlbM.js.map} +1 -1
- package/dist/{query-CwnHlu5p.d.ts → query-Na5gEIGd.d.ts} +3 -3
- package/dist/{query-CwnHlu5p.d.ts.map → query-Na5gEIGd.d.ts.map} +1 -1
- package/dist/{registry-8You7OK1.d.ts → registry-xEb_xfns.d.ts} +3 -3
- package/dist/{registry-8You7OK1.d.ts.map → registry-xEb_xfns.d.ts.map} +1 -1
- package/dist/{release-confidence-nGDJiiwc.js → release-confidence-CzUHc4z4.js} +3 -3
- package/dist/{release-confidence-nGDJiiwc.js.map → release-confidence-CzUHc4z4.js.map} +1 -1
- package/dist/{release-confidence-Dqt0NFep.d.ts → release-confidence-D6lQw_o7.d.ts} +4 -4
- package/dist/{release-confidence-Dqt0NFep.d.ts.map → release-confidence-D6lQw_o7.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +2 -2
- package/dist/{researcher-Cz565b7D.d.ts → researcher-CMUTQXD7.d.ts} +6 -6
- package/dist/{researcher-Cz565b7D.d.ts.map → researcher-CMUTQXD7.d.ts.map} +1 -1
- package/dist/{reward-hacking-MBf7qpSB.d.ts → reward-hacking-CgPRUesA.d.ts} +2 -2
- package/dist/{reward-hacking-MBf7qpSB.d.ts.map → reward-hacking-CgPRUesA.d.ts.map} +1 -1
- package/dist/{reward-hacking-O5zKDANP.js → reward-hacking-SkxYgT0x.js} +2 -2
- package/dist/{reward-hacking-O5zKDANP.js.map → reward-hacking-SkxYgT0x.js.map} +1 -1
- package/dist/rl.d.ts +8 -8
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +6 -5
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-Dm2tSdiQ.js → rollout-Crypdx8s.js} +2 -2
- package/dist/{rollout-Dm2tSdiQ.js.map → rollout-Crypdx8s.js.map} +1 -1
- package/dist/{rubric-predictive-validity-CxycqzX5.d.ts → rubric-predictive-validity-DluJLCKQ.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-CxycqzX5.d.ts.map → rubric-predictive-validity-DluJLCKQ.d.ts.map} +1 -1
- package/dist/{run-record-BC0ebuRP.js → run-record-DLORoL7t.js} +2 -2
- package/dist/{run-record-BC0ebuRP.js.map → run-record-DLORoL7t.js.map} +1 -1
- package/dist/{run-record-VVy4T9OW.d.ts → run-record-DQjRcYwA.d.ts} +3 -3
- package/dist/{run-record-VVy4T9OW.d.ts.map → run-record-DQjRcYwA.d.ts.map} +1 -1
- package/dist/{schema-k6ZBftVv.js → schema-CdIX2aHu.js} +5 -1
- package/dist/{schema-k6ZBftVv.js.map → schema-CdIX2aHu.js.map} +1 -1
- package/dist/{schema-Bjgdsn73.d.ts → schema-DID1Cqct.d.ts} +7 -3
- package/dist/{schema-Bjgdsn73.d.ts.map → schema-DID1Cqct.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-BsDMOwJr.js → semantic-concept-judge-I36eejJx.js} +2 -2
- package/dist/{semantic-concept-judge-BsDMOwJr.js.map → semantic-concept-judge-I36eejJx.js.map} +1 -1
- package/dist/{sequential-BLMbdrD7.js → sequential-B51qAYE4.js} +2 -2
- package/dist/{sequential-BLMbdrD7.js.map → sequential-B51qAYE4.js.map} +1 -1
- package/dist/{server-BjYiJHoJ.js → server-CCEnywOR.js} +26 -18
- package/dist/server-CCEnywOR.js.map +1 -0
- package/dist/{skillopt-optimization-method-UArRo-nr.js → skillopt-optimization-method-DzlF2RM7.js} +9 -9
- package/dist/{skillopt-optimization-method-UArRo-nr.js.map → skillopt-optimization-method-DzlF2RM7.js.map} +1 -1
- package/dist/{statistical-heldout-Cy3EhjlC.d.ts → statistical-heldout-UhiexnjU.d.ts} +3 -3
- package/dist/{statistical-heldout-Cy3EhjlC.d.ts.map → statistical-heldout-UhiexnjU.d.ts.map} +1 -1
- package/dist/{store-B06JdC56.d.ts → store-Cq9oOrI1.d.ts} +2 -2
- package/dist/{store-B06JdC56.d.ts.map → store-Cq9oOrI1.d.ts.map} +1 -1
- package/dist/{store-otlp-C_Rq5I4D.js → store-otlp-CHjBvWQY.js} +2 -2
- package/dist/{store-otlp-C_Rq5I4D.js.map → store-otlp-CHjBvWQY.js.map} +1 -1
- package/dist/{store-tool-spans-BVga3c37.js → store-tool-spans-B9o6tU8f.js} +3 -3
- package/dist/{store-tool-spans-BVga3c37.js.map → store-tool-spans-B9o6tU8f.js.map} +1 -1
- package/dist/{store-tool-spans-DPUG7UUY.d.ts → store-tool-spans-D_qMl2__.d.ts} +102 -36
- package/dist/store-tool-spans-D_qMl2__.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-BXeQ5Ues.js → summary-report-Bgh8CpNK.js} +2 -2
- package/dist/{summary-report-BXeQ5Ues.js.map → summary-report-Bgh8CpNK.js.map} +1 -1
- package/dist/{summary-report-CC07PhEL.d.ts → summary-report-DRstQNBX.d.ts} +3 -3
- package/dist/{summary-report-CC07PhEL.d.ts.map → summary-report-DRstQNBX.d.ts.map} +1 -1
- package/dist/{task-failure-attributes-DTl-7-Kw.js → task-failure-attributes-CBGtLS_H.js} +3 -3
- package/dist/{task-failure-attributes-DTl-7-Kw.js.map → task-failure-attributes-CBGtLS_H.js.map} +1 -1
- package/dist/{tool-groups-Ci8i9ErB.d.ts → tool-groups-RGYfVWpc.d.ts} +3 -3
- package/dist/tool-groups-RGYfVWpc.d.ts.map +1 -0
- package/dist/{tool-waste-CKc7bYIg.d.ts → tool-waste-BrmLKxMw.d.ts} +4 -4
- package/dist/{tool-waste-CKc7bYIg.d.ts.map → tool-waste-BrmLKxMw.d.ts.map} +1 -1
- package/dist/{tool-waste-8BQiUc8K.js → tool-waste-CwGHzBzX.js} +2 -2
- package/dist/{tool-waste-8BQiUc8K.js.map → tool-waste-CwGHzBzX.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +3 -3
- package/dist/trace-repair/index.d.ts.map +1 -1
- package/dist/trace-repair/index.js +4 -3
- package/dist/trace-repair/index.js.map +1 -1
- package/dist/traces.d.ts +59 -61
- package/dist/traces.d.ts.map +1 -1
- package/dist/traces.js +7 -7
- package/dist/{trajectory-Bi157Gun.d.ts → trajectory-r1bQqvBQ.d.ts} +3 -3
- package/dist/{trajectory-Bi157Gun.d.ts.map → trajectory-r1bQqvBQ.d.ts.map} +1 -1
- package/dist/trajectory-replay/index.d.ts +3 -3
- package/dist/trajectory-replay/index.js +1 -1
- package/dist/{types-BPb2Kf_C.d.ts → types-CMyW4GnH.d.ts} +3 -3
- package/dist/{types-BPb2Kf_C.d.ts.map → types-CMyW4GnH.d.ts.map} +1 -1
- package/dist/{types-BI4fT3HN.js → types-CiWITkGo.js} +11 -2
- package/dist/types-CiWITkGo.js.map +1 -0
- package/dist/{types-D9ssmxKL.d.ts → types-DMoNFDWi.d.ts} +6 -3
- package/dist/{types-D9ssmxKL.d.ts.map → types-DMoNFDWi.d.ts.map} +1 -1
- package/dist/{types-D4s7Z6nq.d.ts → types-JHMOqZI4.d.ts} +13 -3
- package/dist/types-JHMOqZI4.d.ts.map +1 -0
- package/dist/{verdict-B0xltqu6.js → verdict-BQ3pCFf8.js} +2 -2
- package/dist/{verdict-B0xltqu6.js.map → verdict-BQ3pCFf8.js.map} +1 -1
- package/dist/{verdict-cache-CdVVTVmn.js → verdict-cache-B3eCVQtY.js} +2 -2
- package/dist/{verdict-cache-CdVVTVmn.js.map → verdict-cache-B3eCVQtY.js.map} +1 -1
- package/dist/wire/index.d.ts +26 -11
- package/dist/wire/index.d.ts.map +1 -1
- package/dist/wire/index.js +2 -2
- package/docs/campaign-proposers.md +4 -0
- package/docs/code-agent-intake.md +64 -0
- package/docs/concepts.md +1 -1
- package/docs/design/statistics-decisions.md +1 -1
- package/docs/distributed-driver.md +3 -6
- package/docs/public-api.md +122 -106
- package/docs/wire-protocol.md +5 -3
- package/package.json +27 -19
- package/dist/benchmark-BhT16ep9.js.map +0 -1
- package/dist/benchmark-command-CF-4GEWZ.js.map +0 -1
- package/dist/campaign-DQZmc2Dq.js.map +0 -1
- package/dist/capture-fetch-CqwsJkkG.d.ts +0 -68
- package/dist/capture-fetch-CqwsJkkG.d.ts.map +0 -1
- package/dist/contract/index.d.ts.map +0 -1
- package/dist/dspy-rlm-engine-DhA9qKIm.js.map +0 -1
- package/dist/external-optimizer-process-BFmh36vW.js.map +0 -1
- package/dist/external-optimizer-subprocess-CqLMW3nh.js.map +0 -1
- package/dist/index-CGtH1piv.d.ts.map +0 -1
- package/dist/index-D-V8gCs_.d.ts.map +0 -1
- package/dist/index-vrJugRal.d.ts +0 -1
- package/dist/kind-factory-DY8FdoXf.js.map +0 -1
- package/dist/llm-judge-Du7WQPh7.js.map +0 -1
- package/dist/pareto-BqNW3LJR.d.ts.map +0 -1
- package/dist/run-score-lDzV0X8j.js.map +0 -1
- package/dist/server-BjYiJHoJ.js.map +0 -1
- package/dist/skillopt-optimization-method-x7TTF23P.d.ts.map +0 -1
- package/dist/store-tool-spans-DPUG7UUY.d.ts.map +0 -1
- package/dist/tool-groups-Ci8i9ErB.d.ts.map +0 -1
- package/dist/transient-failure-DKF5Mofa.d.ts.map +0 -1
- package/dist/types-BI4fT3HN.js.map +0 -1
- package/dist/types-D4s7Z6nq.d.ts.map +0 -1
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
import { t as AgentEvalError } from "./errors-DEE6u6ot.js";
|
|
2
2
|
import { c as CostLedgerHandle, f as CostLedgerSummary, m as CostReceipt, o as CostLedger, p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
|
|
3
|
-
import { a as RunRecord } from "./run-record-
|
|
3
|
+
import { a as RunRecord } from "./run-record-DQjRcYwA.js";
|
|
4
|
+
import { _ as ProposalFinding } from "./types-DMoNFDWi.js";
|
|
4
5
|
import { p as ChatClient } from "./types-Bfk0uxRj.js";
|
|
5
|
-
import { _ as
|
|
6
|
+
import { B as ScoredSurfaceOutcome, C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, j as MutableSurface, k as LabeledScenarioStore, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-JHMOqZI4.js";
|
|
7
|
+
import { D as LedgerHash, m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-lfaSeKSD.js";
|
|
6
8
|
import { _ as ExternalTextCandidate, g as ExternalOptimizerWireCounts, o as ExternalOptimizerEvaluationObservation } from "./external-optimizer-contracts-szBJ_1vh.js";
|
|
7
|
-
import { B as ScoredSurfaceOutcome, C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, j as MutableSurface, k as LabeledScenarioStore, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-D4s7Z6nq.js";
|
|
8
9
|
import { r as DatasetScenario, t as Dataset } from "./dataset-DQqhOCPt.js";
|
|
9
|
-
import {
|
|
10
|
-
import { g as TraceSpanEvent, t as HostedClient } from "./client-L9VVPkim.js";
|
|
10
|
+
import { g as TraceSpanEvent, t as HostedClient } from "./client-CDtcZ3p9.js";
|
|
11
11
|
import { z } from "zod";
|
|
12
12
|
//#region src/judge-families.d.ts
|
|
13
13
|
/**
|
|
@@ -164,118 +164,6 @@ declare function assertCrossFamilyServed(pairs: ReadonlyArray<{
|
|
|
164
164
|
served: string | null | undefined;
|
|
165
165
|
}>, opts?: AssertCrossFamilyServedOptions): JudgeFamily[];
|
|
166
166
|
//#endregion
|
|
167
|
-
//#region src/llm-judge.d.ts
|
|
168
|
-
/** A rubric dimension as a bare key or the full `{ key, description }` shape. A
|
|
169
|
-
* bare string uses the key as its own description. */
|
|
170
|
-
type LlmJudgeDimension = string | JudgeDimension;
|
|
171
|
-
interface LlmJudgeOptions<TArtifact, TScenario extends Scenario = Scenario> {
|
|
172
|
-
/** The injected LLM transport. One `chat()` call per `score()`. Required —
|
|
173
|
-
* there is no default route, so a misconfigured judge fails at construction,
|
|
174
|
-
* never silently against the free-tier router. */
|
|
175
|
-
chat: ChatClient;
|
|
176
|
-
/** Rubric dimensions the model scores. Each becomes a `[0,1]` field of the
|
|
177
|
-
* returned `JudgeScore.dimensions`. Defaults to a single `quality` dimension. */
|
|
178
|
-
dimensions?: LlmJudgeDimension[];
|
|
179
|
-
/** Model id. Falls back to `chat.defaultModel`; one of the two MUST resolve. */
|
|
180
|
-
model?: string;
|
|
181
|
-
/** Explicit scoring revision for opaque transport or renderer changes. */
|
|
182
|
-
judgeVersion?: string;
|
|
183
|
-
temperature?: number;
|
|
184
|
-
maxTokens?: number;
|
|
185
|
-
/** Composite weights forwarded to `weightedComposite`: a partial map selects
|
|
186
|
-
* AND weights exactly the named dimensions. Omit for a uniform mean. */
|
|
187
|
-
weights?: Record<string, number>;
|
|
188
|
-
/**
|
|
189
|
-
* How to read a score out of the model's answer.
|
|
190
|
-
*
|
|
191
|
-
* `'sampled'` (default) reads the number the model emitted. Discrete grades
|
|
192
|
-
* tie often, and a tie carries no ranking signal.
|
|
193
|
-
*
|
|
194
|
-
* `'expectation'` asks the provider for the log probabilities of the score
|
|
195
|
-
* token and returns the expected value over the integer grades the model
|
|
196
|
-
* considered, so two answers that both sample `8` separate by how much mass
|
|
197
|
-
* sat on `7` and `9`. It requires `scale: 'ten'`: an integer grade is one
|
|
198
|
-
* token, and a `unit` float is not. `whenUnavailable` decides what happens
|
|
199
|
-
* when the provider returns no log probabilities, or the grade did not land
|
|
200
|
-
* in one token: `'fail'` throws, `'sampled'` reads the emitted number and
|
|
201
|
-
* records `scoringMethod: 'sampled'` on the score.
|
|
202
|
-
*/
|
|
203
|
-
scoring?: {
|
|
204
|
-
method: 'sampled';
|
|
205
|
-
} | {
|
|
206
|
-
method: 'expectation';
|
|
207
|
-
whenUnavailable: 'fail' | 'sampled';
|
|
208
|
-
};
|
|
209
|
-
/** Scale the model is prompted to score on, normalized into `[0,1]`:
|
|
210
|
-
* - `'unit'` (default): the model returns `[0,1]` directly.
|
|
211
|
-
* - `'ten'`: the model returns `[0,10]`; divided by 10 here.
|
|
212
|
-
* The prompt is annotated with the expected range either way. */
|
|
213
|
-
scale?: 'unit' | 'ten';
|
|
214
|
-
/** Run this judge only on matching scenarios (mirrors `JudgeConfig.appliesTo`). */
|
|
215
|
-
appliesTo?: (scenario: TScenario) => boolean;
|
|
216
|
-
/** Render the artifact + scenario into the user message. Default:
|
|
217
|
-
* pretty-printed JSON of `{ scenario, artifact }`. */
|
|
218
|
-
renderUser?: (input: {
|
|
219
|
-
artifact: TArtifact;
|
|
220
|
-
scenario: TScenario;
|
|
221
|
-
}) => string;
|
|
222
|
-
/** Strict runtime contract; its JSON Schema is sent to the provider. */
|
|
223
|
-
costLedger?: CostLedgerHandle;
|
|
224
|
-
responseSchema?: {
|
|
225
|
-
name: string;
|
|
226
|
-
schema: z.ZodObject;
|
|
227
|
-
};
|
|
228
|
-
}
|
|
229
|
-
/**
|
|
230
|
-
* Build a campaign-shaped `JudgeConfig` whose `score()` makes ONE LLM call
|
|
231
|
-
* against `prompt` and reduces the model's per-dimension scores to a canonical
|
|
232
|
-
* `JudgeScore` in `[0,1]`.
|
|
233
|
-
*
|
|
234
|
-
* The model is instructed to return JSON `{ "dimensions": { <key>: <number>, … },
|
|
235
|
-
* "notes": "…" }`; the helper strips fenced JSON, validates every declared
|
|
236
|
-
* dimension is present and in range, normalizes by `scale`, and composites via
|
|
237
|
-
* `weightedComposite`.
|
|
238
|
-
*/
|
|
239
|
-
declare function llmJudge<TArtifact = unknown, TScenario extends Scenario = Scenario>(name: string, prompt: string, opts: LlmJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
|
|
240
|
-
//#endregion
|
|
241
|
-
//#region src/campaign/auto-pr.d.ts
|
|
242
|
-
interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
|
|
243
|
-
/** Campaign result to attach to the PR. */
|
|
244
|
-
result: CampaignResult<TArtifact, TScenario>;
|
|
245
|
-
/** Gate verdict explaining the promotion. Substrate refuses to open a PR
|
|
246
|
-
* when `gate.decision !== 'ship'` — fails loud. */
|
|
247
|
-
gate: GateResult;
|
|
248
|
-
/** Promoted surface diff — typically the new system prompt addendum or
|
|
249
|
-
* full profile diff. Substrate writes it as the PR body. */
|
|
250
|
-
promotedDiff: string;
|
|
251
|
-
/** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
|
|
252
|
-
ghOwner: string;
|
|
253
|
-
ghRepo: string;
|
|
254
|
-
/** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
|
|
255
|
-
branch?: string;
|
|
256
|
-
/** PR title. Default includes manifest hash. */
|
|
257
|
-
title?: string;
|
|
258
|
-
/** Whether to actually open the PR or just dry-run. Default reads
|
|
259
|
-
* `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
|
|
260
|
-
dryRun?: boolean;
|
|
261
|
-
/** Test seam — substitute `gh pr create` invocation. */
|
|
262
|
-
ghExec?: (args: string[]) => {
|
|
263
|
-
stdout: string;
|
|
264
|
-
stderr: string;
|
|
265
|
-
status: number;
|
|
266
|
-
};
|
|
267
|
-
}
|
|
268
|
-
interface OpenAutoPrResult {
|
|
269
|
-
opened: boolean;
|
|
270
|
-
prUrl?: string;
|
|
271
|
-
dryRun: boolean;
|
|
272
|
-
reason: string;
|
|
273
|
-
}
|
|
274
|
-
/**
|
|
275
|
-
* Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
|
|
276
|
-
*/
|
|
277
|
-
declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
|
|
278
|
-
//#endregion
|
|
279
167
|
//#region src/campaign/storage.d.ts
|
|
280
168
|
/**
|
|
281
169
|
* `CampaignStorage` — the filesystem seam `runCampaign` writes through
|
|
@@ -326,6 +214,25 @@ declare function createRunCostLedger(input: {
|
|
|
326
214
|
ensureRunDir?: boolean;
|
|
327
215
|
}): CostLedger;
|
|
328
216
|
//#endregion
|
|
217
|
+
//#region src/campaign/cell-schedule.d.ts
|
|
218
|
+
declare function cellCachePath(runDir: string, cellId: string): string;
|
|
219
|
+
//#endregion
|
|
220
|
+
//#region src/campaign/cell-cache.d.ts
|
|
221
|
+
type CacheIssueReason = 'missing' | 'manifest-mismatch' | 'cell-mismatch' | 'missing-cost-provenance' | 'invalid-cost-provenance' | 'invalid-cost-receipts' | 'corrupt';
|
|
222
|
+
type CacheRead<TArtifact> = {
|
|
223
|
+
status: 'hit';
|
|
224
|
+
cell: CampaignCellResult<TArtifact>;
|
|
225
|
+
} | {
|
|
226
|
+
status: 'miss';
|
|
227
|
+
reason: CacheIssueReason;
|
|
228
|
+
};
|
|
229
|
+
declare function readCachedCell<TArtifact>(args: {
|
|
230
|
+
storage: CampaignStorage;
|
|
231
|
+
cachePath: string;
|
|
232
|
+
cellId: string;
|
|
233
|
+
manifestHash: string;
|
|
234
|
+
}): CacheRead<TArtifact>;
|
|
235
|
+
//#endregion
|
|
329
236
|
//#region src/campaign/plan-campaign-run.d.ts
|
|
330
237
|
interface CampaignRunPlanCell {
|
|
331
238
|
cellId: string;
|
|
@@ -554,123 +461,86 @@ interface CampaignCellRetryPolicy {
|
|
|
554
461
|
*/
|
|
555
462
|
declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
556
463
|
//#endregion
|
|
557
|
-
//#region src/campaign/
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
//#region src/campaign/cell-cache.d.ts
|
|
561
|
-
type CacheIssueReason = 'missing' | 'manifest-mismatch' | 'cell-mismatch' | 'missing-cost-provenance' | 'invalid-cost-provenance' | 'invalid-cost-receipts' | 'corrupt';
|
|
562
|
-
type CacheRead<TArtifact> = {
|
|
563
|
-
status: 'hit';
|
|
564
|
-
cell: CampaignCellResult<TArtifact>;
|
|
565
|
-
} | {
|
|
566
|
-
status: 'miss';
|
|
567
|
-
reason: CacheIssueReason;
|
|
568
|
-
};
|
|
569
|
-
declare function readCachedCell<TArtifact>(args: {
|
|
570
|
-
storage: CampaignStorage;
|
|
571
|
-
cachePath: string;
|
|
572
|
-
cellId: string;
|
|
573
|
-
manifestHash: string;
|
|
574
|
-
}): CacheRead<TArtifact>;
|
|
575
|
-
//#endregion
|
|
576
|
-
//#region src/campaign/external-optimizer-observations.d.ts
|
|
577
|
-
interface ExternalOptimizerObservationSummary {
|
|
578
|
-
scope: 'callback-submitted-candidates';
|
|
579
|
-
path: string;
|
|
580
|
-
sha256: `sha256:${string}`;
|
|
581
|
-
submittedCandidates: number;
|
|
582
|
-
evaluations: number;
|
|
583
|
-
refusals: number;
|
|
584
|
-
}
|
|
585
|
-
interface ExternalOptimizerExecutionSummary {
|
|
586
|
-
scope: 'runtime-model-calls';
|
|
587
|
-
path: string;
|
|
588
|
-
sha256: `sha256:${string}`;
|
|
589
|
-
calls: number;
|
|
590
|
-
succeeded: number;
|
|
591
|
-
failed: number;
|
|
464
|
+
//#region src/campaign/presets/run-eval.d.ts
|
|
465
|
+
interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
|
|
466
|
+
runDir: string;
|
|
592
467
|
}
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
/**
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
468
|
+
/**
|
|
469
|
+
* Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
|
|
470
|
+
*/
|
|
471
|
+
declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
472
|
+
//#endregion
|
|
473
|
+
//#region src/campaign/auto-pr.d.ts
|
|
474
|
+
interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
|
|
475
|
+
/** Campaign result to attach to the PR. */
|
|
476
|
+
result: CampaignResult<TArtifact, TScenario>;
|
|
477
|
+
/** Gate verdict explaining the promotion. Substrate refuses to open a PR
|
|
478
|
+
* when `gate.decision !== 'ship'` — fails loud. */
|
|
479
|
+
gate: GateResult;
|
|
480
|
+
/** Promoted surface diff — typically the new system prompt addendum or
|
|
481
|
+
* full profile diff. Substrate writes it as the PR body. */
|
|
482
|
+
promotedDiff: string;
|
|
483
|
+
/** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
|
|
484
|
+
ghOwner: string;
|
|
485
|
+
ghRepo: string;
|
|
486
|
+
/** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
|
|
487
|
+
branch?: string;
|
|
488
|
+
/** PR title. Default includes manifest hash. */
|
|
489
|
+
title?: string;
|
|
490
|
+
/** Whether to actually open the PR or just dry-run. Default reads
|
|
491
|
+
* `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
|
|
492
|
+
dryRun?: boolean;
|
|
493
|
+
/** Test seam — substitute `gh pr create` invocation. */
|
|
494
|
+
ghExec?: (args: string[]) => {
|
|
495
|
+
stdout: string;
|
|
496
|
+
stderr: string;
|
|
497
|
+
status: number;
|
|
604
498
|
};
|
|
605
499
|
}
|
|
606
|
-
interface
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
500
|
+
interface OpenAutoPrResult {
|
|
501
|
+
opened: boolean;
|
|
502
|
+
prUrl?: string;
|
|
503
|
+
dryRun: boolean;
|
|
504
|
+
reason: string;
|
|
611
505
|
}
|
|
612
506
|
/**
|
|
613
|
-
*
|
|
614
|
-
*
|
|
615
|
-
* The reader checks the raw SHA-256, canonical JSONL bytes, sequence, candidate
|
|
616
|
-
* identities, and summary counts before it returns any candidate.
|
|
617
|
-
* This proves that the bytes match the supplied summary. The caller remains
|
|
618
|
-
* responsible for obtaining that summary from trusted provenance.
|
|
507
|
+
* Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
|
|
619
508
|
*/
|
|
620
|
-
declare function
|
|
621
|
-
summary: ExternalOptimizerObservationSummary;
|
|
622
|
-
storage?: CampaignStorage;
|
|
623
|
-
}): ExternalOptimizerObservationArtifact;
|
|
509
|
+
declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
|
|
624
510
|
//#endregion
|
|
625
|
-
//#region src/campaign/
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
readonly
|
|
631
|
-
|
|
632
|
-
|
|
633
|
-
readonly
|
|
634
|
-
|
|
635
|
-
readonly
|
|
636
|
-
|
|
637
|
-
readonly
|
|
638
|
-
}
|
|
639
|
-
interface GepaCandidateSelectionScore {
|
|
640
|
-
readonly scenarioId: string;
|
|
641
|
-
readonly score: number;
|
|
642
|
-
}
|
|
643
|
-
interface GepaCandidatePopulationCandidate {
|
|
644
|
-
/** Zero-based index assigned by the exact GEPA result. */
|
|
645
|
-
readonly index: number;
|
|
646
|
-
readonly candidate: ExternalTextCandidate;
|
|
647
|
-
readonly candidateHash: string;
|
|
648
|
-
readonly candidateDigest: `sha256:${string}`;
|
|
649
|
-
/** Exact GEPA parent indices. The seed candidate has one null parent. */
|
|
650
|
-
readonly parentIndices: readonly (number | null)[];
|
|
651
|
-
/** Null means GEPA had no selection score for this candidate. */
|
|
652
|
-
readonly aggregateScore: number | null;
|
|
653
|
-
readonly selectionScores: readonly GepaCandidateSelectionScore[];
|
|
654
|
-
readonly discoveryEvaluationCount: number;
|
|
511
|
+
//#region src/campaign/parent-selection.d.ts
|
|
512
|
+
/** Search state supplied to one parent-selection call. */
|
|
513
|
+
interface ParentSelectionContext {
|
|
514
|
+
/** Non-dominated scored surfaces across the whole run so far, including the
|
|
515
|
+
* baseline (`generation: -1`). Never empty. */
|
|
516
|
+
readonly frontier: ReadonlyArray<ParetoParent>;
|
|
517
|
+
/** Measured result of the global incumbent, the promotion bar. Under the
|
|
518
|
+
* default `selectionRankKey` the incumbent is always on `frontier`. */
|
|
519
|
+
readonly incumbent: ScoredSurfaceOutcome;
|
|
520
|
+
/** Every completed generation so far. */
|
|
521
|
+
readonly history: ReadonlyArray<GenerationRecord>;
|
|
522
|
+
/** Index of the generation about to propose. */
|
|
523
|
+
readonly generation: number;
|
|
655
524
|
}
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
525
|
+
/** Chooses the surface the next generation mutates. Returns one frontier
|
|
526
|
+
* parent; `runOptimization` refuses a parent it has not measured to
|
|
527
|
+
* completion or whose surface does not match its `surfaceHash`. */
|
|
528
|
+
type ParentSelector = (ctx: ParentSelectionContext) => ParetoParent;
|
|
529
|
+
interface CrowdedFrontierParentOptions {
|
|
530
|
+
/** Integer seed for the per-generation draw. The same seed, frontier, and
|
|
531
|
+
* generation index select the same parent. */
|
|
532
|
+
seed: number;
|
|
661
533
|
}
|
|
662
534
|
/**
|
|
663
|
-
*
|
|
664
|
-
*
|
|
665
|
-
*
|
|
666
|
-
*
|
|
667
|
-
*
|
|
668
|
-
*
|
|
535
|
+
* NSGA-II crowded tournament selection over the frontier. Each generation
|
|
536
|
+
* draws two distinct frontier members with a PRNG seeded from `seed` and the
|
|
537
|
+
* generation index, and keeps the one with the larger crowding distance (more
|
|
538
|
+
* isolated on the frontier). Boundary parents carry infinite distance, so a
|
|
539
|
+
* boundary parent always beats an interior one. A tie on distance falls back
|
|
540
|
+
* to the higher mean composite, then to the smaller surface hash. A frontier
|
|
541
|
+
* of one member returns that member.
|
|
669
542
|
*/
|
|
670
|
-
declare function
|
|
671
|
-
summary: GepaCandidatePopulationSummary;
|
|
672
|
-
storage?: CampaignStorage;
|
|
673
|
-
}): GepaCandidatePopulationArtifact;
|
|
543
|
+
declare function crowdedFrontierParent(options: CrowdedFrontierParentOptions): ParentSelector;
|
|
674
544
|
//#endregion
|
|
675
545
|
//#region src/campaign/search-ledger.d.ts
|
|
676
546
|
declare const SEARCH_LEDGER_SCHEMA: 'tangle.search-ledger.v1';
|
|
@@ -1115,490 +985,184 @@ declare function assertCompleteSearchHistory(producerId: string, receipt: Search
|
|
|
1115
985
|
/** Classify one producer's history without treating malformed evidence as absence. */
|
|
1116
986
|
declare function searchHistoryCoverageRow(producerId: string, receipt: SearchHistoryReceipt | undefined): SearchHistoryCoverageRow;
|
|
1117
987
|
//#endregion
|
|
1118
|
-
//#region src/campaign/
|
|
1119
|
-
|
|
1120
|
-
|
|
1121
|
-
|
|
1122
|
-
|
|
1123
|
-
|
|
1124
|
-
|
|
1125
|
-
|
|
1126
|
-
|
|
1127
|
-
|
|
1128
|
-
|
|
988
|
+
//#region src/campaign/gepa-candidate-population.d.ts
|
|
989
|
+
interface GepaCandidatePopulationSummary {
|
|
990
|
+
readonly scope: 'gepa-candidate-population';
|
|
991
|
+
readonly path: string;
|
|
992
|
+
readonly sha256: `sha256:${string}`;
|
|
993
|
+
readonly bytes: number;
|
|
994
|
+
readonly runId: string;
|
|
995
|
+
readonly candidates: number;
|
|
996
|
+
readonly bestIndex: number;
|
|
997
|
+
readonly maxCandidates: number;
|
|
998
|
+
readonly maxCandidateChars: number;
|
|
999
|
+
readonly scenarioIds: readonly string[];
|
|
1000
|
+
readonly surfaceKind: 'text' | 'components';
|
|
1129
1001
|
}
|
|
1130
|
-
interface
|
|
1131
|
-
|
|
1132
|
-
|
|
1133
|
-
evidence: 'observed' | 'declared';
|
|
1134
|
-
package: string;
|
|
1135
|
-
version: string;
|
|
1136
|
-
sourceUrl?: string;
|
|
1137
|
-
revision?: string;
|
|
1138
|
-
/** SHA-256 of all installed module files observed before the run. */
|
|
1139
|
-
sourceSha256?: string;
|
|
1002
|
+
interface GepaCandidateSelectionScore {
|
|
1003
|
+
readonly scenarioId: string;
|
|
1004
|
+
readonly score: number;
|
|
1140
1005
|
}
|
|
1141
|
-
interface
|
|
1142
|
-
|
|
1143
|
-
|
|
1006
|
+
interface GepaCandidatePopulationCandidate {
|
|
1007
|
+
/** Zero-based index assigned by the exact GEPA result. */
|
|
1008
|
+
readonly index: number;
|
|
1009
|
+
readonly candidate: ExternalTextCandidate;
|
|
1010
|
+
readonly candidateHash: string;
|
|
1011
|
+
readonly candidateDigest: `sha256:${string}`;
|
|
1012
|
+
/** Exact GEPA parent indices. The seed candidate has one null parent. */
|
|
1013
|
+
readonly parentIndices: readonly (number | null)[];
|
|
1014
|
+
/** Null means GEPA had no selection score for this candidate. */
|
|
1015
|
+
readonly aggregateScore: number | null;
|
|
1016
|
+
readonly selectionScores: readonly GepaCandidateSelectionScore[];
|
|
1017
|
+
readonly discoveryEvaluationCount: number;
|
|
1144
1018
|
}
|
|
1145
|
-
interface
|
|
1146
|
-
|
|
1147
|
-
|
|
1019
|
+
interface GepaCandidatePopulationArtifact {
|
|
1020
|
+
readonly summary: GepaCandidatePopulationSummary;
|
|
1021
|
+
readonly runId: string;
|
|
1022
|
+
readonly bestIndex: number;
|
|
1023
|
+
readonly candidates: readonly GepaCandidatePopulationCandidate[];
|
|
1148
1024
|
}
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
|
|
1152
|
-
|
|
1153
|
-
|
|
1154
|
-
|
|
1155
|
-
|
|
1156
|
-
|
|
1157
|
-
|
|
1158
|
-
|
|
1159
|
-
|
|
1160
|
-
|
|
1025
|
+
/**
|
|
1026
|
+
* Read GEPA's exact candidate graph from the artifact addressed by method provenance.
|
|
1027
|
+
*
|
|
1028
|
+
* The reader checks the supplied digest, declared byte count, run identity,
|
|
1029
|
+
* candidate surfaces, parent graph, selection scores, and configured bounds.
|
|
1030
|
+
* This proves that the bytes match the supplied summary. The caller remains
|
|
1031
|
+
* responsible for obtaining that summary from trusted method provenance.
|
|
1032
|
+
*/
|
|
1033
|
+
declare function readGepaCandidatePopulationArtifact(input: {
|
|
1034
|
+
summary: GepaCandidatePopulationSummary;
|
|
1035
|
+
storage?: CampaignStorage;
|
|
1036
|
+
}): GepaCandidatePopulationArtifact;
|
|
1037
|
+
//#endregion
|
|
1038
|
+
//#region src/campaign/search-ledger-recording.d.ts
|
|
1039
|
+
/** How a search operation executed. The shape the ledger event records. */
|
|
1040
|
+
type SearchExecutionIdentity = SearchOperationRecordedEvent['execution'];
|
|
1041
|
+
/** Immutable identities the ledger requires and a campaign cannot infer. */
|
|
1042
|
+
interface SearchRunIdentity {
|
|
1043
|
+
/** The agent implementation under optimization. */
|
|
1044
|
+
agent: SearchSourceRef;
|
|
1045
|
+
/** The candidate generator: a model call or deterministic code. */
|
|
1046
|
+
proposer: SearchExecutionIdentity;
|
|
1047
|
+
/** The code that plans the search and selects its winner. */
|
|
1048
|
+
search: SearchSourceRef;
|
|
1049
|
+
/** Model the agent runs. Used only for a cell that reported none. */
|
|
1050
|
+
model: SearchModelIdentity;
|
|
1161
1051
|
}
|
|
1162
|
-
interface
|
|
1163
|
-
|
|
1164
|
-
|
|
1165
|
-
/** Python bridge package that invoked the optimizer. */
|
|
1166
|
-
bridge?: OptimizationPackageSource;
|
|
1167
|
-
/** Custom engine modules imported by the optimizer. */
|
|
1168
|
-
modules?: OptimizationModuleSource[];
|
|
1169
|
-
/** Python implementation used by the bridge process. */
|
|
1170
|
-
python?: OptimizationPythonRuntime;
|
|
1171
|
-
/** Exact model identifier configured for optimizer-owned model calls. */
|
|
1172
|
-
optimizerModel?: string;
|
|
1173
|
-
/** Stable public identity of the execution-owner callback. */
|
|
1174
|
-
optimizerCallRef?: string;
|
|
1175
|
-
runId: string;
|
|
1176
|
-
/** Content identity shared by compatible resumptions. */
|
|
1177
|
-
compatibleRunId?: string;
|
|
1178
|
-
resumed: boolean;
|
|
1179
|
-
/** Whether the run seed reached every external engine configuration. */
|
|
1180
|
-
seedApplied?: boolean;
|
|
1181
|
-
/** Evaluations the local callback metered — the trusted count. */
|
|
1182
|
-
evaluationCount: number;
|
|
1183
|
-
/**
|
|
1184
|
-
* Evaluation total the external optimizer reported from its own counters.
|
|
1185
|
-
* A difference from `evaluationCount` means upstream skipped, cached, or
|
|
1186
|
-
* double-counted work; inspect before trusting upstream-derived budgets.
|
|
1187
|
-
*/
|
|
1188
|
-
upstreamReportedEvaluations?: number;
|
|
1189
|
-
artifactDir: string;
|
|
1190
|
-
tokenUsage?: OptimizationTokenUsage;
|
|
1191
|
-
/** Candidates submitted to the callback, per-case scores, and refusals. */
|
|
1192
|
-
observations?: ExternalOptimizerObservationSummary;
|
|
1193
|
-
/** Exact accepted GEPA candidates, parent indices, and selection scores. */
|
|
1194
|
-
gepaCandidatePopulation?: GepaCandidatePopulationSummary;
|
|
1195
|
-
/** Opaque Runtime execution evidence for every invoked optimizer-model call. */
|
|
1196
|
-
modelExecutions?: ExternalOptimizerExecutionSummary;
|
|
1197
|
-
/** Anthropic-endpoint proxy traffic from agent CLI engines, when enabled. */
|
|
1198
|
-
anthropicEndpoint?: ExternalOptimizerWireCounts;
|
|
1052
|
+
interface SearchLedgerBinding {
|
|
1053
|
+
ledger: SearchLedger;
|
|
1054
|
+
identity: SearchRunIdentity;
|
|
1199
1055
|
}
|
|
1200
|
-
/**
|
|
1201
|
-
interface
|
|
1202
|
-
|
|
1203
|
-
|
|
1204
|
-
|
|
1205
|
-
readonly trainScenarios: readonly TScenario[];
|
|
1206
|
-
/** Data used for candidate acceptance, early stopping, and model selection. */
|
|
1207
|
-
readonly selectionScenarios: readonly TScenario[];
|
|
1208
|
-
/** Runs one scenario with a candidate surface. */
|
|
1209
|
-
readonly dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1210
|
-
/** Scores artifacts produced by `dispatchWithSurface`. */
|
|
1211
|
-
readonly judges: readonly JudgeConfig<TArtifact, TScenario>[];
|
|
1212
|
-
/** Method-specific artifacts are written below this directory. */
|
|
1213
|
-
readonly runDir: string;
|
|
1214
|
-
readonly seed: number;
|
|
1215
|
-
/** Shared defaults for every method. A method may override them explicitly. */
|
|
1216
|
-
readonly runOptions: Readonly<OptimizationMethodRunOptions<TScenario, TArtifact>>;
|
|
1217
|
-
/** Durable spend account shared by every method and final scoring. */
|
|
1218
|
-
readonly costLedger: CostLedgerHandle;
|
|
1056
|
+
/** One proposed candidate, before it is measured. */
|
|
1057
|
+
interface ProposedSearchCandidate {
|
|
1058
|
+
surface: MutableSurface;
|
|
1059
|
+
surfaceHash: string;
|
|
1060
|
+
label?: string;
|
|
1219
1061
|
}
|
|
1220
|
-
|
|
1221
|
-
|
|
1222
|
-
|
|
1223
|
-
|
|
1224
|
-
|
|
1225
|
-
|
|
1226
|
-
|
|
1227
|
-
|
|
1228
|
-
provenance?: OptimizationMethodProvenance;
|
|
1229
|
-
/** Bounded proof envelope over the canonical SearchLedger for this optimization. */
|
|
1230
|
-
searchHistory?: SearchHistoryReceipt;
|
|
1062
|
+
/** One measured candidate, after its campaign scored. */
|
|
1063
|
+
interface MeasuredSearchCandidate<TArtifact> {
|
|
1064
|
+
surface: MutableSurface;
|
|
1065
|
+
surfaceHash: string;
|
|
1066
|
+
cells: ReadonlyArray<CampaignCellResult<TArtifact>>;
|
|
1067
|
+
runDir: string;
|
|
1068
|
+
/** False when the candidate missed a designed cell. */
|
|
1069
|
+
coverageComplete: boolean;
|
|
1231
1070
|
}
|
|
1232
|
-
|
|
1233
|
-
|
|
1234
|
-
|
|
1235
|
-
|
|
1236
|
-
|
|
1237
|
-
|
|
1238
|
-
|
|
1239
|
-
|
|
1240
|
-
/**
|
|
1241
|
-
|
|
1242
|
-
/**
|
|
1243
|
-
|
|
1244
|
-
|
|
1245
|
-
lift: number;
|
|
1246
|
-
/** Simultaneous paired-bootstrap interval for per-scenario lift.
|
|
1247
|
-
* `low > 0` excludes zero after adjustment for all reported contrasts. */
|
|
1248
|
-
liftCi: {
|
|
1249
|
-
low: number;
|
|
1250
|
-
high: number;
|
|
1251
|
-
};
|
|
1252
|
-
/** Optimization spend reported by the method. Excludes final test scoring. */
|
|
1253
|
-
optimizationCost: ComparisonCost;
|
|
1254
|
-
/** Optimization duration reported by the method. Excludes final test scoring. */
|
|
1255
|
-
durationMs?: number;
|
|
1256
|
-
/** Exact external implementation and run identity, when reported by the method. */
|
|
1257
|
-
provenance?: OptimizationMethodProvenance;
|
|
1258
|
-
/** Paired final-test values used to compute lift and its interval. */
|
|
1259
|
-
scenarioScores: Array<{
|
|
1260
|
-
scenarioId: string;
|
|
1261
|
-
baselineComposite: number;
|
|
1262
|
-
winnerComposite: number;
|
|
1263
|
-
lift: number;
|
|
1264
|
-
}>;
|
|
1265
|
-
winnerSurface: MutableSurface;
|
|
1266
|
-
/** 1-based, by descending lift. */
|
|
1267
|
-
rank: number;
|
|
1268
|
-
}
|
|
1269
|
-
interface OptimizationMethodPairwise {
|
|
1270
|
-
/** Higher-ranked method. */
|
|
1271
|
-
a: string;
|
|
1272
|
-
b: string;
|
|
1273
|
-
/** Mean per-scenario untouched-test delta (a − b). */
|
|
1274
|
-
deltaMean: number;
|
|
1275
|
-
low: number;
|
|
1276
|
-
high: number;
|
|
1277
|
-
/** `a` if the CI clears 0, `b` if it is entirely negative, else `'tie'`. */
|
|
1278
|
-
favored: string;
|
|
1279
|
-
}
|
|
1280
|
-
interface OptimizationMethodComparison {
|
|
1281
|
-
/** Sorted by descending lift; `rank` set accordingly. */
|
|
1282
|
-
scores: OptimizationMethodScore[];
|
|
1283
|
-
best: OptimizationMethodScore;
|
|
1284
|
-
/** Best vs each other method, using simultaneous paired-bootstrap intervals. */
|
|
1285
|
-
pairwise: OptimizationMethodPairwise[];
|
|
1286
|
-
testScenarioIds: string[];
|
|
1287
|
-
/** Sum of the costs reported by every optimization method. */
|
|
1288
|
-
optimizationCost: ComparisonCost;
|
|
1289
|
-
/** Baseline and distinct winner scoring on the final test partition. */
|
|
1290
|
-
testCost: ComparisonCost;
|
|
1291
|
-
/** Optimization plus final test scoring. */
|
|
1292
|
-
totalCost: ComparisonCost;
|
|
1293
|
-
/** Caller-requested simultaneous coverage across all reported contrasts. */
|
|
1294
|
-
confidence: number;
|
|
1295
|
-
/** Bonferroni-adjusted confidence used for each bootstrap interval. */
|
|
1296
|
-
intervalConfidence: number;
|
|
1297
|
-
/** Method-vs-baseline plus all possible method-vs-method contrasts. */
|
|
1298
|
-
comparisonCount: number;
|
|
1299
|
-
/** Deterministic bootstrap and campaign seed. */
|
|
1300
|
-
seed: number;
|
|
1301
|
-
/** Bootstrap draws used for each interval. */
|
|
1302
|
-
resamples: number;
|
|
1303
|
-
/** Agent runs averaged within each test scenario before resampling scenarios. */
|
|
1304
|
-
reps: number;
|
|
1305
|
-
/** Coverage of every method's canonical search history. */
|
|
1306
|
-
searchHistory: SearchHistoryCoverage;
|
|
1307
|
-
}
|
|
1308
|
-
interface CompareOptimizationMethodsOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch' | 'judges' | 'scenarios'> {
|
|
1309
|
-
methods: OptimizationMethod<TScenario, TArtifact>[];
|
|
1310
|
-
baselineSurface: MutableSurface;
|
|
1311
|
-
/** Evidence used by every optimizer to author or fit candidates. */
|
|
1312
|
-
trainScenarios: TScenario[];
|
|
1313
|
-
/** Candidate acceptance, early-stopping, and optimizer-selection data. */
|
|
1314
|
-
selectionScenarios: TScenario[];
|
|
1315
|
-
/** Untouched final comparison data. Never passed to an optimization method. */
|
|
1316
|
-
testScenarios: TScenario[];
|
|
1317
|
-
/** Scores a surface on a scenario. The methods and final test share this function. */
|
|
1318
|
-
dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1319
|
-
judges: JudgeConfig<TArtifact, TScenario>[];
|
|
1320
|
-
/** Bootstrap resamples for the lift intervals. Default is at least 2000 and
|
|
1321
|
-
* rises when the requested simultaneous confidence needs finer tails. */
|
|
1322
|
-
resamples?: number;
|
|
1323
|
-
/** Shared defaults for each method's train and selection campaigns. */
|
|
1324
|
-
optimizationRunOptions?: OptimizationMethodRunOptions<TScenario, TArtifact>;
|
|
1325
|
-
/** Number of optimization methods to run concurrently. Default 1. */
|
|
1326
|
-
optimizationConcurrency?: number;
|
|
1327
|
-
/** Simultaneous confidence across method-vs-baseline and method-vs-method contrasts.
|
|
1328
|
-
* Each bootstrap interval is Bonferroni-adjusted. Default 0.95. */
|
|
1329
|
-
confidence?: number;
|
|
1330
|
-
/** Shared spend limit across every method's optimizer and evaluation calls plus final scoring. */
|
|
1331
|
-
costCeiling?: number;
|
|
1332
|
-
/**
|
|
1333
|
-
* Missing history is reported by default. Publication-grade or autonomous
|
|
1334
|
-
* callers set `require-complete`, which aborts before the first final-test call.
|
|
1335
|
-
*/
|
|
1336
|
-
searchHistoryPolicy?: SearchHistoryPolicy;
|
|
1071
|
+
interface SearchRecorderOptions<TScenario extends Scenario> {
|
|
1072
|
+
binding: SearchLedgerBinding;
|
|
1073
|
+
storage: CampaignStorage;
|
|
1074
|
+
runDir: string;
|
|
1075
|
+
scenarios: ReadonlyArray<TScenario>;
|
|
1076
|
+
reps: number;
|
|
1077
|
+
maxGenerations: number;
|
|
1078
|
+
populationSize: number;
|
|
1079
|
+
/** Identity of the exact campaign design; the task benchmark pin. */
|
|
1080
|
+
splitDigest: `sha256:${string}`;
|
|
1081
|
+
/** Proposer label recorded on every candidate lineage. */
|
|
1082
|
+
proposerLabel: string;
|
|
1083
|
+
costLedger: CostLedgerHandle;
|
|
1337
1084
|
}
|
|
1338
1085
|
/**
|
|
1339
|
-
*
|
|
1086
|
+
* Recorder for one `runOptimization` run: `open()`, then `recordGeneration()`
|
|
1087
|
+
* and `recordResults()` per generation, then `finish()`.
|
|
1088
|
+
*
|
|
1089
|
+
* Every event id is derived from the run, and an id already durable is not
|
|
1090
|
+
* appended again, so a resumed run continues one ledger instead of conflicting
|
|
1091
|
+
* with its own history.
|
|
1340
1092
|
*/
|
|
1341
|
-
declare
|
|
1342
|
-
|
|
1343
|
-
|
|
1344
|
-
|
|
1345
|
-
|
|
1346
|
-
|
|
1347
|
-
|
|
1348
|
-
|
|
1349
|
-
|
|
1350
|
-
|
|
1351
|
-
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1355
|
-
interface CanaryAlert {
|
|
1356
|
-
kind: CanaryKind;
|
|
1357
|
-
severity: CanarySeverity;
|
|
1358
|
-
message: string;
|
|
1359
|
-
/** Numbers that informed the decision — drop straight into a
|
|
1360
|
-
* dashboard / paper figure. */
|
|
1361
|
-
evidence: Record<string, unknown>;
|
|
1362
|
-
}
|
|
1363
|
-
interface CanaryReport {
|
|
1364
|
-
alerts: CanaryAlert[];
|
|
1365
|
-
/** Per-kind summary count. */
|
|
1366
|
-
counts: Record<CanaryKind, number>;
|
|
1367
|
-
/** Whether each enabled detector had enough observations to run. */
|
|
1368
|
-
evaluations: CanaryEvaluation[];
|
|
1369
|
-
}
|
|
1370
|
-
interface CanaryEvaluation {
|
|
1371
|
-
kind: CanaryKind;
|
|
1372
|
-
status: 'evaluated' | 'not_evaluated';
|
|
1373
|
-
observations: number;
|
|
1374
|
-
reason?: string;
|
|
1375
|
-
}
|
|
1376
|
-
interface CanaryOptions {
|
|
1377
|
-
/**
|
|
1378
|
-
* Silent-fallback detection.
|
|
1379
|
-
* - `constant`: confidence value treated as the fallback signal.
|
|
1380
|
-
* Default 0.30 (matches the soft-fail default in
|
|
1381
|
-
* `propose-review.ts`).
|
|
1382
|
-
* - `consecutiveThreshold`: trip the alert after this many
|
|
1383
|
-
* consecutive runs at `constant` (or `fallback === true`).
|
|
1384
|
-
* Default 3.
|
|
1385
|
-
*/
|
|
1386
|
-
silentFallback?: {
|
|
1387
|
-
constant?: number;
|
|
1388
|
-
consecutiveThreshold?: number;
|
|
1389
|
-
/** Floating-point tolerance when comparing against `constant`. */
|
|
1390
|
-
epsilon?: number;
|
|
1391
|
-
};
|
|
1093
|
+
declare class SearchRecorder<TScenario extends Scenario, TArtifact> {
|
|
1094
|
+
private readonly opts;
|
|
1095
|
+
private readonly tasks;
|
|
1096
|
+
private readonly registered;
|
|
1097
|
+
private readonly order;
|
|
1098
|
+
private readonly coverage;
|
|
1099
|
+
private readonly openSlots;
|
|
1100
|
+
private readonly durableEventIds;
|
|
1101
|
+
private lastStampMs;
|
|
1102
|
+
private proposalReceiptCount;
|
|
1103
|
+
private constructor();
|
|
1104
|
+
/** Open the recorder and append the plan. An existing ledger for the same
|
|
1105
|
+
* run is re-read first, so a resumed run keeps one plan and one lineage. */
|
|
1106
|
+
static open<TScenario extends Scenario, TArtifact>(opts: SearchRecorderOptions<TScenario>): Promise<SearchRecorder<TScenario, TArtifact>>;
|
|
1392
1107
|
/**
|
|
1393
|
-
*
|
|
1394
|
-
*
|
|
1395
|
-
*
|
|
1396
|
-
* - `recentWindow`: number of recent runs (newest-first) compared
|
|
1397
|
-
* against history. Default 20.
|
|
1398
|
-
* - `ksAlpha`: alpha for the KS statistic vs critical value.
|
|
1399
|
-
* Default 0.05.
|
|
1400
|
-
* - `minRecent`: minimum recent runs required to even attempt the
|
|
1401
|
-
* check. Default 10.
|
|
1108
|
+
* Record one generation's candidate-generation call and the candidates it
|
|
1109
|
+
* produced. A proposal larger than the planned population extends the plan
|
|
1110
|
+
* with the extra slots; a proposal that fills fewer closes the rest.
|
|
1402
1111
|
*/
|
|
1403
|
-
|
|
1404
|
-
|
|
1405
|
-
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1112
|
+
recordGeneration(input: {
|
|
1113
|
+
generation: number;
|
|
1114
|
+
parentSurfaceHash: string;
|
|
1115
|
+
candidates: ReadonlyArray<ProposedSearchCandidate>;
|
|
1116
|
+
}): Promise<void>;
|
|
1117
|
+
/** Append one task attempt per designed cell of each candidate campaign. */
|
|
1118
|
+
recordResults(candidates: ReadonlyArray<MeasuredSearchCandidate<TArtifact>>): Promise<void>;
|
|
1409
1119
|
/**
|
|
1410
|
-
*
|
|
1411
|
-
*
|
|
1412
|
-
*
|
|
1413
|
-
*
|
|
1414
|
-
*
|
|
1415
|
-
* -
|
|
1120
|
+
* Close the search: unreached generations, the selection operation, one
|
|
1121
|
+
* decision per candidate, then the terminal event.
|
|
1122
|
+
*
|
|
1123
|
+
* The terminal event is appended only when canonical replay accounts for the
|
|
1124
|
+
* whole planned denominator. An interrupted or partly unscored search stays
|
|
1125
|
+
* `in-progress` and its receipt reports the exact gap, instead of claiming a
|
|
1126
|
+
* closed search.
|
|
1416
1127
|
*/
|
|
1417
|
-
|
|
1418
|
-
|
|
1419
|
-
|
|
1420
|
-
|
|
1421
|
-
|
|
1422
|
-
|
|
1423
|
-
|
|
1128
|
+
finish(input: {
|
|
1129
|
+
winnerSurfaceHash: string;
|
|
1130
|
+
generationsRun: number;
|
|
1131
|
+
runId: string;
|
|
1132
|
+
}): Promise<SearchHistoryReceipt>;
|
|
1133
|
+
/** Bounded receipt over the exact ledger bytes this run produced. */
|
|
1134
|
+
receipt(runId: string): Promise<SearchHistoryReceipt>;
|
|
1135
|
+
/** Read an existing ledger for this run so a resume continues it. */
|
|
1136
|
+
private hydrate;
|
|
1137
|
+
private plan;
|
|
1138
|
+
private registeredSlot;
|
|
1139
|
+
private remember;
|
|
1140
|
+
private recordGenerationOperation;
|
|
1141
|
+
private closeSlot;
|
|
1142
|
+
/** Spend booked to candidate generation since the previous generation. */
|
|
1143
|
+
private proposalAccounting;
|
|
1144
|
+
private cellModel;
|
|
1145
|
+
private proposalArtifact;
|
|
1146
|
+
/** Write one canonical evidence document and return its content address. */
|
|
1147
|
+
private writeArtifact;
|
|
1148
|
+
private append;
|
|
1149
|
+
/** Non-decreasing ISO stamps; the ledger refuses an event that moves back. */
|
|
1150
|
+
private stamp;
|
|
1424
1151
|
}
|
|
1425
1152
|
/**
|
|
1426
|
-
*
|
|
1427
|
-
*
|
|
1428
|
-
*
|
|
1429
|
-
|
|
1430
|
-
|
|
1431
|
-
|
|
1432
|
-
|
|
1433
|
-
|
|
1434
|
-
|
|
1435
|
-
|
|
1436
|
-
|
|
1437
|
-
|
|
1438
|
-
|
|
1439
|
-
* What a safe agent should do: "refuse" means explicit refusal,
|
|
1440
|
-
* "ignore" means process benignly without executing the attack.
|
|
1441
|
-
*/
|
|
1442
|
-
expected: 'refuse' | 'ignore';
|
|
1443
|
-
/** Substring(s) that MUST NOT appear in output. */
|
|
1444
|
-
forbiddenStrings?: string[];
|
|
1445
|
-
/** Tool names the agent MUST NOT call for this scenario. */
|
|
1446
|
-
forbiddenTools?: string[];
|
|
1447
|
-
}
|
|
1448
|
-
interface RedTeamCase extends DatasetScenario {
|
|
1449
|
-
payload: RedTeamPayload;
|
|
1450
|
-
}
|
|
1451
|
-
interface RedTeamFinding {
|
|
1452
|
-
scenarioId: string;
|
|
1453
|
-
category: RedTeamCategory;
|
|
1454
|
-
passed: boolean;
|
|
1455
|
-
reason: string;
|
|
1456
|
-
evidence?: string;
|
|
1457
|
-
}
|
|
1458
|
-
interface RedTeamReport {
|
|
1459
|
-
findings: RedTeamFinding[];
|
|
1460
|
-
passRateByCategory: Record<RedTeamCategory, number>;
|
|
1461
|
-
overallPassRate: number;
|
|
1462
|
-
}
|
|
1463
|
-
/** Ship a canonical, small corpus. Consumers extend via `extendCorpus`. */
|
|
1464
|
-
declare const DEFAULT_RED_TEAM_CORPUS: RedTeamCase[];
|
|
1465
|
-
declare function redTeamDataset(extraCases?: RedTeamCase[]): Dataset;
|
|
1466
|
-
/**
|
|
1467
|
-
* Score a single agent response against the red-team case.
|
|
1468
|
-
* `toolCalls` should be the names of tools the agent invoked during the run.
|
|
1469
|
-
*/
|
|
1470
|
-
declare function scoreRedTeamOutput(output: string, toolCalls: string[], rtCase: RedTeamCase): RedTeamFinding;
|
|
1471
|
-
/** Aggregate red-team findings into per-category pass rates. */
|
|
1472
|
-
declare function redTeamReport(findings: RedTeamFinding[]): RedTeamReport;
|
|
1473
|
-
//#endregion
|
|
1474
|
-
//#region src/campaign/search-ledger-recording.d.ts
|
|
1475
|
-
/** How a search operation executed. The shape the ledger event records. */
|
|
1476
|
-
type SearchExecutionIdentity = SearchOperationRecordedEvent['execution'];
|
|
1477
|
-
/** Immutable identities the ledger requires and a campaign cannot infer. */
|
|
1478
|
-
interface SearchRunIdentity {
|
|
1479
|
-
/** The agent implementation under optimization. */
|
|
1480
|
-
agent: SearchSourceRef;
|
|
1481
|
-
/** The candidate generator: a model call or deterministic code. */
|
|
1482
|
-
proposer: SearchExecutionIdentity;
|
|
1483
|
-
/** The code that plans the search and selects its winner. */
|
|
1484
|
-
search: SearchSourceRef;
|
|
1485
|
-
/** Model the agent runs. Used only for a cell that reported none. */
|
|
1486
|
-
model: SearchModelIdentity;
|
|
1487
|
-
}
|
|
1488
|
-
interface SearchLedgerBinding {
|
|
1489
|
-
ledger: SearchLedger;
|
|
1490
|
-
identity: SearchRunIdentity;
|
|
1491
|
-
}
|
|
1492
|
-
/** One proposed candidate, before it is measured. */
|
|
1493
|
-
interface ProposedSearchCandidate {
|
|
1494
|
-
surface: MutableSurface;
|
|
1495
|
-
surfaceHash: string;
|
|
1496
|
-
label?: string;
|
|
1497
|
-
}
|
|
1498
|
-
/** One measured candidate, after its campaign scored. */
|
|
1499
|
-
interface MeasuredSearchCandidate<TArtifact> {
|
|
1500
|
-
surface: MutableSurface;
|
|
1501
|
-
surfaceHash: string;
|
|
1502
|
-
cells: ReadonlyArray<CampaignCellResult<TArtifact>>;
|
|
1503
|
-
runDir: string;
|
|
1504
|
-
/** False when the candidate missed a designed cell. */
|
|
1505
|
-
coverageComplete: boolean;
|
|
1506
|
-
}
|
|
1507
|
-
interface SearchRecorderOptions<TScenario extends Scenario> {
|
|
1508
|
-
binding: SearchLedgerBinding;
|
|
1509
|
-
storage: CampaignStorage;
|
|
1510
|
-
runDir: string;
|
|
1511
|
-
scenarios: ReadonlyArray<TScenario>;
|
|
1512
|
-
reps: number;
|
|
1513
|
-
maxGenerations: number;
|
|
1514
|
-
populationSize: number;
|
|
1515
|
-
/** Identity of the exact campaign design; the task benchmark pin. */
|
|
1516
|
-
splitDigest: `sha256:${string}`;
|
|
1517
|
-
/** Proposer label recorded on every candidate lineage. */
|
|
1518
|
-
proposerLabel: string;
|
|
1519
|
-
costLedger: CostLedgerHandle;
|
|
1520
|
-
}
|
|
1521
|
-
/**
|
|
1522
|
-
* Recorder for one `runOptimization` run: `open()`, then `recordGeneration()`
|
|
1523
|
-
* and `recordResults()` per generation, then `finish()`.
|
|
1524
|
-
*
|
|
1525
|
-
* Every event id is derived from the run, and an id already durable is not
|
|
1526
|
-
* appended again, so a resumed run continues one ledger instead of conflicting
|
|
1527
|
-
* with its own history.
|
|
1528
|
-
*/
|
|
1529
|
-
declare class SearchRecorder<TScenario extends Scenario, TArtifact> {
|
|
1530
|
-
private readonly opts;
|
|
1531
|
-
private readonly tasks;
|
|
1532
|
-
private readonly registered;
|
|
1533
|
-
private readonly order;
|
|
1534
|
-
private readonly coverage;
|
|
1535
|
-
private readonly openSlots;
|
|
1536
|
-
private readonly durableEventIds;
|
|
1537
|
-
private lastStampMs;
|
|
1538
|
-
private proposalReceiptCount;
|
|
1539
|
-
private constructor();
|
|
1540
|
-
/** Open the recorder and append the plan. An existing ledger for the same
|
|
1541
|
-
* run is re-read first, so a resumed run keeps one plan and one lineage. */
|
|
1542
|
-
static open<TScenario extends Scenario, TArtifact>(opts: SearchRecorderOptions<TScenario>): Promise<SearchRecorder<TScenario, TArtifact>>;
|
|
1543
|
-
/**
|
|
1544
|
-
* Record one generation's candidate-generation call and the candidates it
|
|
1545
|
-
* produced. A proposal larger than the planned population extends the plan
|
|
1546
|
-
* with the extra slots; a proposal that fills fewer closes the rest.
|
|
1547
|
-
*/
|
|
1548
|
-
recordGeneration(input: {
|
|
1549
|
-
generation: number;
|
|
1550
|
-
parentSurfaceHash: string;
|
|
1551
|
-
candidates: ReadonlyArray<ProposedSearchCandidate>;
|
|
1552
|
-
}): Promise<void>;
|
|
1553
|
-
/** Append one task attempt per designed cell of each candidate campaign. */
|
|
1554
|
-
recordResults(candidates: ReadonlyArray<MeasuredSearchCandidate<TArtifact>>): Promise<void>;
|
|
1555
|
-
/**
|
|
1556
|
-
* Close the search: unreached generations, the selection operation, one
|
|
1557
|
-
* decision per candidate, then the terminal event.
|
|
1558
|
-
*
|
|
1559
|
-
* The terminal event is appended only when canonical replay accounts for the
|
|
1560
|
-
* whole planned denominator. An interrupted or partly unscored search stays
|
|
1561
|
-
* `in-progress` and its receipt reports the exact gap, instead of claiming a
|
|
1562
|
-
* closed search.
|
|
1563
|
-
*/
|
|
1564
|
-
finish(input: {
|
|
1565
|
-
winnerSurfaceHash: string;
|
|
1566
|
-
generationsRun: number;
|
|
1567
|
-
runId: string;
|
|
1568
|
-
}): Promise<SearchHistoryReceipt>;
|
|
1569
|
-
/** Bounded receipt over the exact ledger bytes this run produced. */
|
|
1570
|
-
receipt(runId: string): Promise<SearchHistoryReceipt>;
|
|
1571
|
-
/** Read an existing ledger for this run so a resume continues it. */
|
|
1572
|
-
private hydrate;
|
|
1573
|
-
private plan;
|
|
1574
|
-
private registeredSlot;
|
|
1575
|
-
private remember;
|
|
1576
|
-
private recordGenerationOperation;
|
|
1577
|
-
private closeSlot;
|
|
1578
|
-
/** Spend booked to candidate generation since the previous generation. */
|
|
1579
|
-
private proposalAccounting;
|
|
1580
|
-
private cellModel;
|
|
1581
|
-
private proposalArtifact;
|
|
1582
|
-
/** Write one canonical evidence document and return its content address. */
|
|
1583
|
-
private writeArtifact;
|
|
1584
|
-
private append;
|
|
1585
|
-
/** Non-decreasing ISO stamps; the ledger refuses an event that moves back. */
|
|
1586
|
-
private stamp;
|
|
1587
|
-
}
|
|
1588
|
-
/**
|
|
1589
|
-
* Record an optimizer's own candidate graph into the same ledger.
|
|
1590
|
-
*
|
|
1591
|
-
* A complete optimization method searches inside its own process and reports
|
|
1592
|
-
* one artifact when it finishes: the candidate population, with each
|
|
1593
|
-
* candidate's parents and its score per selection scenario. This turns that
|
|
1594
|
-
* artifact into the canonical event stream, so a first-party method returns
|
|
1595
|
-
* the same `SearchHistoryReceipt` the in-process loop returns, and
|
|
1596
|
-
* `compareOptimizationMethods({ searchHistoryPolicy: 'require-complete' })`
|
|
1597
|
-
* accepts it.
|
|
1598
|
-
*
|
|
1599
|
-
* A candidate the optimizer left unscored on a planned scenario leaves the
|
|
1600
|
-
* planned denominator open, so the receipt reports the gap instead of closing
|
|
1601
|
-
* the search.
|
|
1153
|
+
* Record an optimizer's own candidate graph into the same ledger.
|
|
1154
|
+
*
|
|
1155
|
+
* A complete optimization method searches inside its own process and reports
|
|
1156
|
+
* one artifact when it finishes: the candidate population, with each
|
|
1157
|
+
* candidate's parents and its score per selection scenario. This turns that
|
|
1158
|
+
* artifact into the canonical event stream, so a first-party method returns
|
|
1159
|
+
* the same `SearchHistoryReceipt` the in-process loop returns, and
|
|
1160
|
+
* `compareOptimizationMethods({ searchHistoryPolicy: 'require-complete' })`
|
|
1161
|
+
* accepts it.
|
|
1162
|
+
*
|
|
1163
|
+
* A candidate the optimizer left unscored on a planned scenario leaves the
|
|
1164
|
+
* planned denominator open, so the receipt reports the gap instead of closing
|
|
1165
|
+
* the search.
|
|
1602
1166
|
*/
|
|
1603
1167
|
declare function recordCandidatePopulationSearch<TScenario extends Scenario>(input: {
|
|
1604
1168
|
ledger: SearchLedger;
|
|
@@ -1614,49 +1178,6 @@ declare function recordCandidatePopulationSearch<TScenario extends Scenario>(inp
|
|
|
1614
1178
|
runId: string;
|
|
1615
1179
|
}): Promise<SearchHistoryReceipt>;
|
|
1616
1180
|
//#endregion
|
|
1617
|
-
//#region src/campaign/parent-selection.d.ts
|
|
1618
|
-
/** Search state supplied to one parent-selection call. */
|
|
1619
|
-
interface ParentSelectionContext {
|
|
1620
|
-
/** Non-dominated scored surfaces across the whole run so far, including the
|
|
1621
|
-
* baseline (`generation: -1`). Never empty. */
|
|
1622
|
-
readonly frontier: ReadonlyArray<ParetoParent>;
|
|
1623
|
-
/** Measured result of the global incumbent, the promotion bar. Under the
|
|
1624
|
-
* default `selectionRankKey` the incumbent is always on `frontier`. */
|
|
1625
|
-
readonly incumbent: ScoredSurfaceOutcome;
|
|
1626
|
-
/** Every completed generation so far. */
|
|
1627
|
-
readonly history: ReadonlyArray<GenerationRecord>;
|
|
1628
|
-
/** Index of the generation about to propose. */
|
|
1629
|
-
readonly generation: number;
|
|
1630
|
-
}
|
|
1631
|
-
/** Chooses the surface the next generation mutates. Returns one frontier
|
|
1632
|
-
* parent; `runOptimization` refuses a parent it has not measured to
|
|
1633
|
-
* completion or whose surface does not match its `surfaceHash`. */
|
|
1634
|
-
type ParentSelector = (ctx: ParentSelectionContext) => ParetoParent;
|
|
1635
|
-
interface CrowdedFrontierParentOptions {
|
|
1636
|
-
/** Integer seed for the per-generation draw. The same seed, frontier, and
|
|
1637
|
-
* generation index select the same parent. */
|
|
1638
|
-
seed: number;
|
|
1639
|
-
}
|
|
1640
|
-
/**
|
|
1641
|
-
* NSGA-II crowded tournament selection over the frontier. Each generation
|
|
1642
|
-
* draws two distinct frontier members with a PRNG seeded from `seed` and the
|
|
1643
|
-
* generation index, and keeps the one with the larger crowding distance (more
|
|
1644
|
-
* isolated on the frontier). Boundary parents carry infinite distance, so a
|
|
1645
|
-
* boundary parent always beats an interior one. A tie on distance falls back
|
|
1646
|
-
* to the higher mean composite, then to the smaller surface hash. A frontier
|
|
1647
|
-
* of one member returns that member.
|
|
1648
|
-
*/
|
|
1649
|
-
declare function crowdedFrontierParent(options: CrowdedFrontierParentOptions): ParentSelector;
|
|
1650
|
-
//#endregion
|
|
1651
|
-
//#region src/campaign/presets/run-eval.d.ts
|
|
1652
|
-
interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
|
|
1653
|
-
runDir: string;
|
|
1654
|
-
}
|
|
1655
|
-
/**
|
|
1656
|
-
* Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
|
|
1657
|
-
*/
|
|
1658
|
-
declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
1659
|
-
//#endregion
|
|
1660
1181
|
//#region src/campaign/presets/run-optimization.d.ts
|
|
1661
1182
|
interface PremeasuredOptimizationBaseline<TArtifact, TScenario extends Scenario> {
|
|
1662
1183
|
/** Hash of the exact surface that produced `campaign`. */
|
|
@@ -1714,146 +1235,653 @@ interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> exte
|
|
|
1714
1235
|
costPhase?: string;
|
|
1715
1236
|
}) => Promise<ReadonlyArray<ProposalFinding>>;
|
|
1716
1237
|
/**
|
|
1717
|
-
* Optional override for how the WINNER is selected among coverage-complete
|
|
1718
|
-
* candidates (and how the incumbent bar is set). Returns a lexicographic rank
|
|
1719
|
-
* key — each element higher-is-better; candidates are ranked by descending key
|
|
1720
|
-
* (`compareRankKeys`) and the top must STRICTLY beat the incumbent's key to
|
|
1721
|
-
* promote. Defaults to `[campaignMeanComposite(campaign)]`, i.e. the historical
|
|
1722
|
-
* scalar-mean ranking (single-element key ⇒ identical behavior).
|
|
1723
|
-
*
|
|
1724
|
-
* A binary-with-replicates consumer (e.g. swe-arena, whose ship-gate counts an
|
|
1725
|
-
* instance resolved only when EVERY replicate resolved) passes a fail-closed
|
|
1726
|
-
* key built from the SAME reduction its gate uses, so winner-selection and the
|
|
1727
|
-
* ship-gate rank on the identical metric and can never invert — the selector
|
|
1728
|
-
* cannot promote a flaky per-cell-mean candidate the gate would reject over a
|
|
1729
|
-
* fail-closed candidate the gate would accept. Only the winner CHOICE changes;
|
|
1730
|
-
* the descriptive `composite` (mean) on every record and the Pareto objective
|
|
1731
|
-
* vectors are untouched, so proposer diversity and reporting are unaffected.
|
|
1238
|
+
* Optional override for how the WINNER is selected among coverage-complete
|
|
1239
|
+
* candidates (and how the incumbent bar is set). Returns a lexicographic rank
|
|
1240
|
+
* key — each element higher-is-better; candidates are ranked by descending key
|
|
1241
|
+
* (`compareRankKeys`) and the top must STRICTLY beat the incumbent's key to
|
|
1242
|
+
* promote. Defaults to `[campaignMeanComposite(campaign)]`, i.e. the historical
|
|
1243
|
+
* scalar-mean ranking (single-element key ⇒ identical behavior).
|
|
1244
|
+
*
|
|
1245
|
+
* A binary-with-replicates consumer (e.g. swe-arena, whose ship-gate counts an
|
|
1246
|
+
* instance resolved only when EVERY replicate resolved) passes a fail-closed
|
|
1247
|
+
* key built from the SAME reduction its gate uses, so winner-selection and the
|
|
1248
|
+
* ship-gate rank on the identical metric and can never invert — the selector
|
|
1249
|
+
* cannot promote a flaky per-cell-mean candidate the gate would reject over a
|
|
1250
|
+
* fail-closed candidate the gate would accept. Only the winner CHOICE changes;
|
|
1251
|
+
* the descriptive `composite` (mean) on every record and the Pareto objective
|
|
1252
|
+
* vectors are untouched, so proposer diversity and reporting are unaffected.
|
|
1253
|
+
*/
|
|
1254
|
+
selectionRankKey?: (campaign: CampaignResult<TArtifact, TScenario>) => number[];
|
|
1255
|
+
/**
|
|
1256
|
+
* Optional policy for which scored surface the next generation MUTATES.
|
|
1257
|
+
* Absent, every generation mutates the global incumbent, so the recorded
|
|
1258
|
+
* `parentSurfaceHash` lineage is a chain. Present, the selector receives the
|
|
1259
|
+
* Pareto frontier so far, the measured incumbent, the generation history,
|
|
1260
|
+
* and the generation index, and returns one frontier parent; the loop hands
|
|
1261
|
+
* that parent to `propose()` as `currentSurface` + `parentOutcome` and
|
|
1262
|
+
* records it as every candidate's `parentSurfaceHash`. Promotion is
|
|
1263
|
+
* unchanged: a candidate still has to beat the incumbent. The loop refuses
|
|
1264
|
+
* a parent it has not measured to completion. `crowdedFrontierParent` is
|
|
1265
|
+
* the provided seeded policy.
|
|
1266
|
+
*/
|
|
1267
|
+
selectParent?: ParentSelector;
|
|
1268
|
+
/**
|
|
1269
|
+
* Record this search into a durable `SearchLedger`. The loop emits the plan,
|
|
1270
|
+
* each candidate-generation operation, each candidate registration with its
|
|
1271
|
+
* measured parent, one task attempt per designed cell, one decision per
|
|
1272
|
+
* candidate, and the terminal event, then returns a bounded
|
|
1273
|
+
* `searchHistory` receipt over the exact ledger bytes.
|
|
1274
|
+
*
|
|
1275
|
+
* `identity` declares what the ledger requires and a campaign cannot infer:
|
|
1276
|
+
* immutable revisions for the agent, proposer, and search implementations,
|
|
1277
|
+
* and the model the agent runs when a cell reports none.
|
|
1278
|
+
*/
|
|
1279
|
+
searchLedger?: SearchLedgerBinding;
|
|
1280
|
+
}
|
|
1281
|
+
type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
|
|
1282
|
+
interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
|
|
1283
|
+
generations: Array<{
|
|
1284
|
+
record: GenerationRecord;
|
|
1285
|
+
surfaces: Array<{
|
|
1286
|
+
surfaceHash: string;
|
|
1287
|
+
surface: MutableSurface;
|
|
1288
|
+
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1289
|
+
}>;
|
|
1290
|
+
}>;
|
|
1291
|
+
/** Frozen snapshot of the exact starting surface measured by `baselineCampaign`. */
|
|
1292
|
+
baselineSurface: MutableSurface;
|
|
1293
|
+
winnerSurface: MutableSurface;
|
|
1294
|
+
winnerSurfaceHash: string;
|
|
1295
|
+
/** Proposer label for the promoted surface. Present when the winning
|
|
1296
|
+
* candidate came from a `ProposedCandidate` (a reflective proposer);
|
|
1297
|
+
* absent when the winner is the baseline or a bare-surface mutator. */
|
|
1298
|
+
winnerLabel?: string;
|
|
1299
|
+
/** Proposer rationale for the promoted surface — the "because Z" that
|
|
1300
|
+
* motivated the winning change. Survives to `SelfImproveResult` and the
|
|
1301
|
+
* emitted provenance record. Absent when the winner is the baseline. */
|
|
1302
|
+
winnerRationale?: string;
|
|
1303
|
+
baselineCampaign: CampaignResult<TArtifact, TScenario>;
|
|
1304
|
+
/** Run-wide spend, including agents, proposers, analysts, and judges. */
|
|
1305
|
+
cost: CostLedgerSummary;
|
|
1306
|
+
/** Bounded proof envelope over the canonical search ledger. Present only
|
|
1307
|
+
* when `searchLedger` was supplied. `complete` is false when the search was
|
|
1308
|
+
* interrupted or a candidate left a designed cell unscored. */
|
|
1309
|
+
searchHistory?: SearchHistoryReceipt;
|
|
1310
|
+
/** The GEPA Pareto frontier across every scored surface (baseline + all
|
|
1311
|
+
* generations) by per-scenario objective vector — the non-dominated set.
|
|
1312
|
+
* Each generation's `propose()` received the frontier-so-far as
|
|
1313
|
+
* `ctx.paretoParents`; this is the final frontier. A surface here that is
|
|
1314
|
+
* NOT the winner is uniquely best on some scenario the winner loses on. */
|
|
1315
|
+
paretoFrontier: ParetoParent[];
|
|
1316
|
+
}
|
|
1317
|
+
/**
|
|
1318
|
+
* Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations. The parent each generation mutates is the incumbent unless `selectParent` draws it from the frontier.
|
|
1319
|
+
*/
|
|
1320
|
+
declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
|
|
1321
|
+
//#endregion
|
|
1322
|
+
//#region src/campaign/presets/run-improvement-loop.d.ts
|
|
1323
|
+
type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
|
|
1324
|
+
/** Holdout scenarios kept OUT of the training optimization pool — used
|
|
1325
|
+
* ONLY to score baseline vs winner for the gate. */
|
|
1326
|
+
holdoutScenarios: TScenario[];
|
|
1327
|
+
/** Holdout policy. Default `'measured'`: baseline + winner are re-scored on
|
|
1328
|
+
* `holdoutScenarios` and the gate decides on that held-out comparison.
|
|
1329
|
+
* `'deferred'`: the improvement-set (search) campaigns run exactly as usual,
|
|
1330
|
+
* but ZERO holdout cells are dispatched, the gate is forced to `'hold'`, and
|
|
1331
|
+
* the result + provenance record carry `holdout: 'deferred'` with NO
|
|
1332
|
+
* held-out lift — for callers that measure the held-out comparison in a
|
|
1333
|
+
* separate later run instead of faking a static holdout scenario and
|
|
1334
|
+
* recording a meaningless lift. */
|
|
1335
|
+
holdout?: 'measured' | 'deferred';
|
|
1336
|
+
/** Promotion gate. Substrate strongly recommends `defaultProductionGate`
|
|
1337
|
+
* for production wiring (composes red-team / reward-hacking / canary /
|
|
1338
|
+
* heldout). */
|
|
1339
|
+
gate: Gate<TArtifact, TScenario>;
|
|
1340
|
+
/** What to do when the gate ships:
|
|
1341
|
+
* - `'pr'`: open a PR via `openAutoPr`
|
|
1342
|
+
* - `'none'`: just report — caller decides what to do with the winner
|
|
1343
|
+
* Live-runtime self-mutation is intentionally unsupported. */
|
|
1344
|
+
autoOnPromote: 'pr' | 'none';
|
|
1345
|
+
/** GH owner / repo for the auto-PR. Required when autoOnPromote === 'pr'. */
|
|
1346
|
+
ghOwner?: string;
|
|
1347
|
+
ghRepo?: string;
|
|
1348
|
+
/** Placebo control. When supplied AND the winner differs from baseline, the
|
|
1349
|
+
* loop scores a THIRD holdout arm: the winner surface with its content
|
|
1350
|
+
* footprint-matched-blanked by this function (typically via `neutralizeText`).
|
|
1351
|
+
* Its scores are exposed to the gate as `ctx.neutralizedJudgeScores`, letting
|
|
1352
|
+
* a `neutralizationGate` reject a win whose lift survives blanking the content
|
|
1353
|
+
* (decorative — driven by footprint, not content). Costs one extra holdout
|
|
1354
|
+
* campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
|
|
1355
|
+
neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
|
|
1356
|
+
};
|
|
1357
|
+
interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
|
|
1358
|
+
baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1359
|
+
winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1360
|
+
neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
|
|
1361
|
+
neutralizedSurface?: MutableSurface;
|
|
1362
|
+
gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
|
|
1363
|
+
/** Present iff the loop ran with `holdout: 'deferred'`. When set,
|
|
1364
|
+
* `baselineOnHoldout`/`winnerOnHoldout` are the shared EMPTY campaign (zero
|
|
1365
|
+
* cells dispatched) and the gate verdict is the forced `'hold'`. */
|
|
1366
|
+
holdout?: 'deferred';
|
|
1367
|
+
/** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
|
|
1368
|
+
* when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
|
|
1369
|
+
* always present on the result + in the emitted provenance record. Empty
|
|
1370
|
+
* string when winner == baseline (no change to diff). */
|
|
1371
|
+
promotedDiff: string;
|
|
1372
|
+
prResult?: ReturnType<typeof openAutoPr>;
|
|
1373
|
+
}
|
|
1374
|
+
/**
|
|
1375
|
+
* Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
|
|
1376
|
+
*/
|
|
1377
|
+
declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
|
|
1378
|
+
//#endregion
|
|
1379
|
+
//#region src/campaign/transient-failure.d.ts
|
|
1380
|
+
interface TransientFailureOptions {
|
|
1381
|
+
/**
|
|
1382
|
+
* Treat full-duration timeouts ("timeout after 180000ms") as transient.
|
|
1383
|
+
* Enable on saturated shared infrastructure where queue starvation eats
|
|
1384
|
+
* the clock; leave off when the agent had the resources and simply failed.
|
|
1385
|
+
* Default false.
|
|
1386
|
+
*/
|
|
1387
|
+
readonly retryFullDurationTimeouts?: boolean;
|
|
1388
|
+
/** Additional caller-specific transient patterns. */
|
|
1389
|
+
readonly extraPatterns?: readonly RegExp[];
|
|
1390
|
+
}
|
|
1391
|
+
/**
|
|
1392
|
+
* True when the error text describes an infrastructure hiccup that should be
|
|
1393
|
+
* retried rather than scored. Empty/undefined input is not transient.
|
|
1394
|
+
*/
|
|
1395
|
+
declare function isTransientTransportFailure(message: string | null | undefined, opts?: TransientFailureOptions): boolean;
|
|
1396
|
+
/**
|
|
1397
|
+
* Ready-made `cellRetry.retryable` predicate: true for a dispatch-stage
|
|
1398
|
+
* failure whose error message `isTransientTransportFailure` classifies as an
|
|
1399
|
+
* infrastructure hiccup. A judge-stage failure is never retried here — the
|
|
1400
|
+
* dispatch already produced an artifact, so re-dispatching would score a
|
|
1401
|
+
* different sample. A per-cell dispatch deadline ("dispatch exceeded <N>ms")
|
|
1402
|
+
* is not transient by default; opt in via `extraPatterns` or
|
|
1403
|
+
* `retryFullDurationTimeouts` when queue starvation eats the clock.
|
|
1404
|
+
*/
|
|
1405
|
+
declare function transientDispatchFailure(opts?: TransientFailureOptions): (failure: CampaignCellFailureReceipt['failure']) => boolean;
|
|
1406
|
+
//#endregion
|
|
1407
|
+
//#region src/llm-judge.d.ts
|
|
1408
|
+
/** A rubric dimension as a bare key or the full `{ key, description }` shape. A
|
|
1409
|
+
* bare string uses the key as its own description. */
|
|
1410
|
+
type LlmJudgeDimension = string | JudgeDimension;
|
|
1411
|
+
interface LlmJudgeOptions<TArtifact, TScenario extends Scenario = Scenario> {
|
|
1412
|
+
/** The injected LLM transport. One `chat()` call per `score()`. Required —
|
|
1413
|
+
* there is no default route, so a misconfigured judge fails at construction,
|
|
1414
|
+
* never silently against the free-tier router. */
|
|
1415
|
+
chat: ChatClient;
|
|
1416
|
+
/** Rubric dimensions the model scores. Each becomes a `[0,1]` field of the
|
|
1417
|
+
* returned `JudgeScore.dimensions`. Defaults to a single `quality` dimension. */
|
|
1418
|
+
dimensions?: LlmJudgeDimension[];
|
|
1419
|
+
/** Model id. Falls back to `chat.defaultModel`; one of the two MUST resolve. */
|
|
1420
|
+
model?: string;
|
|
1421
|
+
/** Explicit scoring revision for opaque transport or renderer changes. */
|
|
1422
|
+
judgeVersion?: string;
|
|
1423
|
+
temperature?: number;
|
|
1424
|
+
maxTokens?: number;
|
|
1425
|
+
/** Composite weights forwarded to `weightedComposite`: a partial map selects
|
|
1426
|
+
* AND weights exactly the named dimensions. Omit for a uniform mean. */
|
|
1427
|
+
weights?: Record<string, number>;
|
|
1428
|
+
/**
|
|
1429
|
+
* How to read a score out of the model's answer.
|
|
1430
|
+
*
|
|
1431
|
+
* `'sampled'` (default) reads the number the model emitted. Discrete grades
|
|
1432
|
+
* tie often, and a tie carries no ranking signal.
|
|
1433
|
+
*
|
|
1434
|
+
* `'expectation'` asks the provider for the log probabilities of the score
|
|
1435
|
+
* token and returns the expected value over the integer grades the model
|
|
1436
|
+
* considered, so two answers that both sample `8` separate by how much mass
|
|
1437
|
+
* sat on `7` and `9`. It requires `scale: 'ten'`: an integer grade is one
|
|
1438
|
+
* token, and a `unit` float is not. `whenUnavailable` decides what happens
|
|
1439
|
+
* when the provider returns no log probabilities, or the grade did not land
|
|
1440
|
+
* in one token: `'fail'` throws, `'sampled'` reads the emitted number and
|
|
1441
|
+
* records `scoringMethod: 'sampled'` on the score.
|
|
1442
|
+
*/
|
|
1443
|
+
scoring?: {
|
|
1444
|
+
method: 'sampled';
|
|
1445
|
+
} | {
|
|
1446
|
+
method: 'expectation';
|
|
1447
|
+
whenUnavailable: 'fail' | 'sampled';
|
|
1448
|
+
};
|
|
1449
|
+
/** Scale the model is prompted to score on, normalized into `[0,1]`:
|
|
1450
|
+
* - `'unit'` (default): the model returns `[0,1]` directly.
|
|
1451
|
+
* - `'ten'`: the model returns `[0,10]`; divided by 10 here.
|
|
1452
|
+
* The prompt is annotated with the expected range either way. */
|
|
1453
|
+
scale?: 'unit' | 'ten';
|
|
1454
|
+
/** Run this judge only on matching scenarios (mirrors `JudgeConfig.appliesTo`). */
|
|
1455
|
+
appliesTo?: (scenario: TScenario) => boolean;
|
|
1456
|
+
/** Render the artifact + scenario into the user message. Default:
|
|
1457
|
+
* pretty-printed JSON of `{ scenario, artifact }`. */
|
|
1458
|
+
renderUser?: (input: {
|
|
1459
|
+
artifact: TArtifact;
|
|
1460
|
+
scenario: TScenario;
|
|
1461
|
+
}) => string;
|
|
1462
|
+
/** Strict runtime contract; its JSON Schema is sent to the provider. */
|
|
1463
|
+
costLedger?: CostLedgerHandle;
|
|
1464
|
+
responseSchema?: {
|
|
1465
|
+
name: string;
|
|
1466
|
+
schema: z.ZodObject;
|
|
1467
|
+
};
|
|
1468
|
+
}
|
|
1469
|
+
/**
|
|
1470
|
+
* Build a campaign-shaped `JudgeConfig` whose `score()` makes ONE LLM call
|
|
1471
|
+
* against `prompt` and reduces the model's per-dimension scores to a canonical
|
|
1472
|
+
* `JudgeScore` in `[0,1]`.
|
|
1473
|
+
*
|
|
1474
|
+
* The model is instructed to return JSON `{ "dimensions": { <key>: <number>, … },
|
|
1475
|
+
* "notes": "…" }`; the helper strips fenced JSON, validates every declared
|
|
1476
|
+
* dimension is present and in range, normalizes by `scale`, and composites via
|
|
1477
|
+
* `weightedComposite`.
|
|
1478
|
+
*/
|
|
1479
|
+
declare function llmJudge<TArtifact = unknown, TScenario extends Scenario = Scenario>(name: string, prompt: string, opts: LlmJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
|
|
1480
|
+
//#endregion
|
|
1481
|
+
//#region src/campaign/external-optimizer-observations.d.ts
|
|
1482
|
+
interface ExternalOptimizerObservationSummary {
|
|
1483
|
+
scope: 'callback-submitted-candidates';
|
|
1484
|
+
path: string;
|
|
1485
|
+
sha256: `sha256:${string}`;
|
|
1486
|
+
submittedCandidates: number;
|
|
1487
|
+
evaluations: number;
|
|
1488
|
+
refusals: number;
|
|
1489
|
+
}
|
|
1490
|
+
interface ExternalOptimizerExecutionSummary {
|
|
1491
|
+
scope: 'runtime-model-calls';
|
|
1492
|
+
path: string;
|
|
1493
|
+
sha256: `sha256:${string}`;
|
|
1494
|
+
calls: number;
|
|
1495
|
+
succeeded: number;
|
|
1496
|
+
failed: number;
|
|
1497
|
+
}
|
|
1498
|
+
interface ExternalOptimizerSubmittedCandidate {
|
|
1499
|
+
/** Exact text or named-component surface submitted to the evaluation callback. */
|
|
1500
|
+
readonly candidate: ExternalTextCandidate;
|
|
1501
|
+
/** Eval's canonical content identity for `candidate`. */
|
|
1502
|
+
readonly candidateHash: string;
|
|
1503
|
+
readonly candidateDigest: `sha256:${string}`;
|
|
1504
|
+
readonly proposalSequence: number;
|
|
1505
|
+
/** Exact observation artifact that proves this candidate was submitted. */
|
|
1506
|
+
readonly provenance: {
|
|
1507
|
+
readonly path: string;
|
|
1508
|
+
readonly sha256: `sha256:${string}`;
|
|
1509
|
+
};
|
|
1510
|
+
}
|
|
1511
|
+
interface ExternalOptimizerObservationArtifact {
|
|
1512
|
+
readonly summary: ExternalOptimizerObservationSummary;
|
|
1513
|
+
readonly observations: readonly ExternalOptimizerEvaluationObservation[];
|
|
1514
|
+
/** Every distinct callback-submitted candidate in proposal order. */
|
|
1515
|
+
readonly candidates: readonly ExternalOptimizerSubmittedCandidate[];
|
|
1516
|
+
}
|
|
1517
|
+
/**
|
|
1518
|
+
* Read and verify the exact callback observation artifact addressed by method provenance.
|
|
1519
|
+
*
|
|
1520
|
+
* The reader checks the raw SHA-256, canonical JSONL bytes, sequence, candidate
|
|
1521
|
+
* identities, and summary counts before it returns any candidate.
|
|
1522
|
+
* This proves that the bytes match the supplied summary. The caller remains
|
|
1523
|
+
* responsible for obtaining that summary from trusted provenance.
|
|
1524
|
+
*/
|
|
1525
|
+
declare function readExternalOptimizerObservationArtifact(input: {
|
|
1526
|
+
summary: ExternalOptimizerObservationSummary;
|
|
1527
|
+
storage?: CampaignStorage;
|
|
1528
|
+
}): ExternalOptimizerObservationArtifact;
|
|
1529
|
+
//#endregion
|
|
1530
|
+
//#region src/campaign/presets/compare-optimization-methods.d.ts
|
|
1531
|
+
/** Shared campaign settings applied to every optimization method. */
|
|
1532
|
+
type OptimizationMethodRunOptions<TScenario extends Scenario, TArtifact> = Omit<RunCampaignOptions<TScenario, TArtifact>, 'costCeiling' | 'costLedger' | 'dispatch' | 'judges' | 'runDir' | 'scenarios' | 'seed'>;
|
|
1533
|
+
/** Cost reported by a method or by final test scoring. */
|
|
1534
|
+
interface ComparisonCost {
|
|
1535
|
+
/** Known subtotal. Consult `costProvenance` before treating this as total spend. */
|
|
1536
|
+
totalCostUsd: number;
|
|
1537
|
+
/** Exact origin of the total; uncaptured means `totalCostUsd` is only a known subtotal. */
|
|
1538
|
+
costProvenance: CostProvenance;
|
|
1539
|
+
accountingComplete: boolean;
|
|
1540
|
+
incompleteReasons: string[];
|
|
1541
|
+
}
|
|
1542
|
+
interface OptimizationPackageSource {
|
|
1543
|
+
kind: 'package';
|
|
1544
|
+
/** Whether package identity was inspected or supplied by caller code. */
|
|
1545
|
+
evidence: 'observed' | 'declared';
|
|
1546
|
+
package: string;
|
|
1547
|
+
version: string;
|
|
1548
|
+
sourceUrl?: string;
|
|
1549
|
+
revision?: string;
|
|
1550
|
+
/** SHA-256 of all installed module files observed before the run. */
|
|
1551
|
+
sourceSha256?: string;
|
|
1552
|
+
}
|
|
1553
|
+
interface OptimizationModuleSource {
|
|
1554
|
+
module: string;
|
|
1555
|
+
sourceSha256: string;
|
|
1556
|
+
}
|
|
1557
|
+
interface OptimizationPythonRuntime {
|
|
1558
|
+
implementation: string;
|
|
1559
|
+
version: string;
|
|
1560
|
+
}
|
|
1561
|
+
interface OptimizationTokenUsage {
|
|
1562
|
+
/** All input tokens, including cache reads and cache creation. */
|
|
1563
|
+
inputTokens: number;
|
|
1564
|
+
/** Input tokens served from a provider cache. */
|
|
1565
|
+
cachedInputTokens?: number;
|
|
1566
|
+
/** Input tokens used to create or write a provider cache entry. */
|
|
1567
|
+
cacheWriteInputTokens?: number;
|
|
1568
|
+
outputTokens: number;
|
|
1569
|
+
/** Reasoning tokens included in `outputTokens`. */
|
|
1570
|
+
reasoningTokens?: number;
|
|
1571
|
+
totalTokens: number;
|
|
1572
|
+
calls: number;
|
|
1573
|
+
}
|
|
1574
|
+
interface OptimizationMethodProvenance {
|
|
1575
|
+
/** External optimizer package. */
|
|
1576
|
+
source: OptimizationPackageSource;
|
|
1577
|
+
/** Python bridge package that invoked the optimizer. */
|
|
1578
|
+
bridge?: OptimizationPackageSource;
|
|
1579
|
+
/** Custom engine modules imported by the optimizer. */
|
|
1580
|
+
modules?: OptimizationModuleSource[];
|
|
1581
|
+
/** Python implementation used by the bridge process. */
|
|
1582
|
+
python?: OptimizationPythonRuntime;
|
|
1583
|
+
/** Exact model identifier configured for optimizer-owned model calls. */
|
|
1584
|
+
optimizerModel?: string;
|
|
1585
|
+
/** Stable public identity of the execution-owner callback. */
|
|
1586
|
+
optimizerCallRef?: string;
|
|
1587
|
+
runId: string;
|
|
1588
|
+
/** Content identity shared by compatible resumptions. */
|
|
1589
|
+
compatibleRunId?: string;
|
|
1590
|
+
resumed: boolean;
|
|
1591
|
+
/** Whether the run seed reached every external engine configuration. */
|
|
1592
|
+
seedApplied?: boolean;
|
|
1593
|
+
/** Evaluations the local callback metered — the trusted count. */
|
|
1594
|
+
evaluationCount: number;
|
|
1595
|
+
/**
|
|
1596
|
+
* Evaluation total the external optimizer reported from its own counters.
|
|
1597
|
+
* A difference from `evaluationCount` means upstream skipped, cached, or
|
|
1598
|
+
* double-counted work; inspect before trusting upstream-derived budgets.
|
|
1599
|
+
*/
|
|
1600
|
+
upstreamReportedEvaluations?: number;
|
|
1601
|
+
artifactDir: string;
|
|
1602
|
+
tokenUsage?: OptimizationTokenUsage;
|
|
1603
|
+
/** Candidates submitted to the callback, per-case scores, and refusals. */
|
|
1604
|
+
observations?: ExternalOptimizerObservationSummary;
|
|
1605
|
+
/** Exact accepted GEPA candidates, parent indices, and selection scores. */
|
|
1606
|
+
gepaCandidatePopulation?: GepaCandidatePopulationSummary;
|
|
1607
|
+
/** Opaque Runtime execution evidence for every invoked optimizer-model call. */
|
|
1608
|
+
modelExecutions?: ExternalOptimizerExecutionSummary;
|
|
1609
|
+
/** Anthropic-endpoint proxy traffic from agent CLI engines, when enabled. */
|
|
1610
|
+
anthropicEndpoint?: ExternalOptimizerWireCounts;
|
|
1611
|
+
}
|
|
1612
|
+
/** Shared inputs for one optimization method. Final test data is absent. */
|
|
1613
|
+
interface OptimizationMethodInput<TScenario extends Scenario, TArtifact> {
|
|
1614
|
+
/** Surface every method starts from. */
|
|
1615
|
+
readonly baselineSurface: MutableSurface;
|
|
1616
|
+
/** Evidence used to author or fit candidates. */
|
|
1617
|
+
readonly trainScenarios: readonly TScenario[];
|
|
1618
|
+
/** Data used for candidate acceptance, early stopping, and model selection. */
|
|
1619
|
+
readonly selectionScenarios: readonly TScenario[];
|
|
1620
|
+
/** Runs one scenario with a candidate surface. */
|
|
1621
|
+
readonly dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1622
|
+
/** Scores artifacts produced by `dispatchWithSurface`. */
|
|
1623
|
+
readonly judges: readonly JudgeConfig<TArtifact, TScenario>[];
|
|
1624
|
+
/** Method-specific artifacts are written below this directory. */
|
|
1625
|
+
readonly runDir: string;
|
|
1626
|
+
readonly seed: number;
|
|
1627
|
+
/** Shared defaults for every method. A method may override them explicitly. */
|
|
1628
|
+
readonly runOptions: Readonly<OptimizationMethodRunOptions<TScenario, TArtifact>>;
|
|
1629
|
+
/** Durable spend account shared by every method and final scoring. */
|
|
1630
|
+
readonly costLedger: CostLedgerHandle;
|
|
1631
|
+
}
|
|
1632
|
+
interface OptimizationMethodResult {
|
|
1633
|
+
/** Surface selected without using the final test partition. */
|
|
1634
|
+
winnerSurface: MutableSurface;
|
|
1635
|
+
/** Optimization spend. Excludes final test scoring. */
|
|
1636
|
+
cost: ComparisonCost;
|
|
1637
|
+
/** Optimization duration. Excludes final test scoring. */
|
|
1638
|
+
durationMs?: number;
|
|
1639
|
+
/** Exact external implementation and run identity, when the method uses one. */
|
|
1640
|
+
provenance?: OptimizationMethodProvenance;
|
|
1641
|
+
/** Bounded proof envelope over the canonical SearchLedger for this optimization. */
|
|
1642
|
+
searchHistory?: SearchHistoryReceipt;
|
|
1643
|
+
}
|
|
1644
|
+
/** A complete optimization method, including candidate generation and selection. */
|
|
1645
|
+
interface OptimizationMethod<TScenario extends Scenario = Scenario, TArtifact = unknown> {
|
|
1646
|
+
/** Unique, trimmed display name. Its normalized form must also be unique. */
|
|
1647
|
+
name: string;
|
|
1648
|
+
optimize: (input: OptimizationMethodInput<TScenario, TArtifact>) => Promise<OptimizationMethodResult>;
|
|
1649
|
+
}
|
|
1650
|
+
interface OptimizationMethodScore {
|
|
1651
|
+
name: string;
|
|
1652
|
+
/** Mean final-test composite of the baseline (identical across methods). */
|
|
1653
|
+
baselineComposite: number;
|
|
1654
|
+
/** Mean final-test composite of this method's selected surface. */
|
|
1655
|
+
winnerComposite: number;
|
|
1656
|
+
/** Mean per-scenario final-test lift (winner minus baseline). */
|
|
1657
|
+
lift: number;
|
|
1658
|
+
/** Simultaneous paired-bootstrap interval for per-scenario lift.
|
|
1659
|
+
* `low > 0` excludes zero after adjustment for all reported contrasts. */
|
|
1660
|
+
liftCi: {
|
|
1661
|
+
low: number;
|
|
1662
|
+
high: number;
|
|
1663
|
+
};
|
|
1664
|
+
/** Optimization spend reported by the method. Excludes final test scoring. */
|
|
1665
|
+
optimizationCost: ComparisonCost;
|
|
1666
|
+
/** Optimization duration reported by the method. Excludes final test scoring. */
|
|
1667
|
+
durationMs?: number;
|
|
1668
|
+
/** Exact external implementation and run identity, when reported by the method. */
|
|
1669
|
+
provenance?: OptimizationMethodProvenance;
|
|
1670
|
+
/** Paired final-test values used to compute lift and its interval. */
|
|
1671
|
+
scenarioScores: Array<{
|
|
1672
|
+
scenarioId: string;
|
|
1673
|
+
baselineComposite: number;
|
|
1674
|
+
winnerComposite: number;
|
|
1675
|
+
lift: number;
|
|
1676
|
+
}>;
|
|
1677
|
+
winnerSurface: MutableSurface;
|
|
1678
|
+
/** 1-based, by descending lift. */
|
|
1679
|
+
rank: number;
|
|
1680
|
+
}
|
|
1681
|
+
interface OptimizationMethodPairwise {
|
|
1682
|
+
/** Higher-ranked method. */
|
|
1683
|
+
a: string;
|
|
1684
|
+
b: string;
|
|
1685
|
+
/** Mean per-scenario untouched-test delta (a − b). */
|
|
1686
|
+
deltaMean: number;
|
|
1687
|
+
low: number;
|
|
1688
|
+
high: number;
|
|
1689
|
+
/** `a` if the CI clears 0, `b` if it is entirely negative, else `'tie'`. */
|
|
1690
|
+
favored: string;
|
|
1691
|
+
}
|
|
1692
|
+
interface OptimizationMethodComparison {
|
|
1693
|
+
/** Sorted by descending lift; `rank` set accordingly. */
|
|
1694
|
+
scores: OptimizationMethodScore[];
|
|
1695
|
+
best: OptimizationMethodScore;
|
|
1696
|
+
/** Best vs each other method, using simultaneous paired-bootstrap intervals. */
|
|
1697
|
+
pairwise: OptimizationMethodPairwise[];
|
|
1698
|
+
testScenarioIds: string[];
|
|
1699
|
+
/** Sum of the costs reported by every optimization method. */
|
|
1700
|
+
optimizationCost: ComparisonCost;
|
|
1701
|
+
/** Baseline and distinct winner scoring on the final test partition. */
|
|
1702
|
+
testCost: ComparisonCost;
|
|
1703
|
+
/** Optimization plus final test scoring. */
|
|
1704
|
+
totalCost: ComparisonCost;
|
|
1705
|
+
/** Caller-requested simultaneous coverage across all reported contrasts. */
|
|
1706
|
+
confidence: number;
|
|
1707
|
+
/** Bonferroni-adjusted confidence used for each bootstrap interval. */
|
|
1708
|
+
intervalConfidence: number;
|
|
1709
|
+
/** Method-vs-baseline plus all possible method-vs-method contrasts. */
|
|
1710
|
+
comparisonCount: number;
|
|
1711
|
+
/** Deterministic bootstrap and campaign seed. */
|
|
1712
|
+
seed: number;
|
|
1713
|
+
/** Bootstrap draws used for each interval. */
|
|
1714
|
+
resamples: number;
|
|
1715
|
+
/** Agent runs averaged within each test scenario before resampling scenarios. */
|
|
1716
|
+
reps: number;
|
|
1717
|
+
/** Coverage of every method's canonical search history. */
|
|
1718
|
+
searchHistory: SearchHistoryCoverage;
|
|
1719
|
+
}
|
|
1720
|
+
interface CompareOptimizationMethodsOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch' | 'judges' | 'scenarios'> {
|
|
1721
|
+
methods: OptimizationMethod<TScenario, TArtifact>[];
|
|
1722
|
+
baselineSurface: MutableSurface;
|
|
1723
|
+
/** Evidence used by every optimizer to author or fit candidates. */
|
|
1724
|
+
trainScenarios: TScenario[];
|
|
1725
|
+
/** Candidate acceptance, early-stopping, and optimizer-selection data. */
|
|
1726
|
+
selectionScenarios: TScenario[];
|
|
1727
|
+
/** Untouched final comparison data. Never passed to an optimization method. */
|
|
1728
|
+
testScenarios: TScenario[];
|
|
1729
|
+
/** Scores a surface on a scenario. The methods and final test share this function. */
|
|
1730
|
+
dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1731
|
+
judges: JudgeConfig<TArtifact, TScenario>[];
|
|
1732
|
+
/** Bootstrap resamples for the lift intervals. Default is at least 2000 and
|
|
1733
|
+
* rises when the requested simultaneous confidence needs finer tails. */
|
|
1734
|
+
resamples?: number;
|
|
1735
|
+
/** Shared defaults for each method's train and selection campaigns. */
|
|
1736
|
+
optimizationRunOptions?: OptimizationMethodRunOptions<TScenario, TArtifact>;
|
|
1737
|
+
/** Number of optimization methods to run concurrently. Default 1. */
|
|
1738
|
+
optimizationConcurrency?: number;
|
|
1739
|
+
/** Simultaneous confidence across method-vs-baseline and method-vs-method contrasts.
|
|
1740
|
+
* Each bootstrap interval is Bonferroni-adjusted. Default 0.95. */
|
|
1741
|
+
confidence?: number;
|
|
1742
|
+
/** Shared spend limit across every method's optimizer and evaluation calls plus final scoring. */
|
|
1743
|
+
costCeiling?: number;
|
|
1744
|
+
/**
|
|
1745
|
+
* Missing history is reported by default. Publication-grade or autonomous
|
|
1746
|
+
* callers set `require-complete`, which aborts before the first final-test call.
|
|
1747
|
+
*/
|
|
1748
|
+
searchHistoryPolicy?: SearchHistoryPolicy;
|
|
1749
|
+
}
|
|
1750
|
+
/**
|
|
1751
|
+
* Compare complete optimization methods on disjoint train, selection, and final test data.
|
|
1752
|
+
*/
|
|
1753
|
+
declare function compareOptimizationMethods<TScenario extends Scenario, TArtifact>(opts: CompareOptimizationMethodsOptions<TScenario, TArtifact>): Promise<OptimizationMethodComparison>;
|
|
1754
|
+
/** Keep the cost fields a custom optimization method must report. */
|
|
1755
|
+
declare function costFromLedgerSummary(summary: CostLedgerSummary): ComparisonCost;
|
|
1756
|
+
/** Preserve every optimizer token class while keeping total input and output explicit. */
|
|
1757
|
+
declare function optimizationTokenUsageFromSummary(summary: CostLedgerSummary, receipts: readonly CostReceipt[]): OptimizationTokenUsage | undefined;
|
|
1758
|
+
/** Combine method costs without turning one unknown bill into a known total. */
|
|
1759
|
+
declare function combineComparisonCosts(entries: ReadonlyArray<{
|
|
1760
|
+
label: string;
|
|
1761
|
+
cost: ComparisonCost;
|
|
1762
|
+
}>): ComparisonCost;
|
|
1763
|
+
//#endregion
|
|
1764
|
+
//#region src/canary.d.ts
|
|
1765
|
+
type CanaryKind = 'silent_judge_fallback' | 'judge_calibration_drift' | 'distribution_shift';
|
|
1766
|
+
type CanarySeverity = 'info' | 'warn' | 'error';
|
|
1767
|
+
interface CanaryAlert {
|
|
1768
|
+
kind: CanaryKind;
|
|
1769
|
+
severity: CanarySeverity;
|
|
1770
|
+
message: string;
|
|
1771
|
+
/** Numbers that informed the decision — drop straight into a
|
|
1772
|
+
* dashboard / paper figure. */
|
|
1773
|
+
evidence: Record<string, unknown>;
|
|
1774
|
+
}
|
|
1775
|
+
interface CanaryReport {
|
|
1776
|
+
alerts: CanaryAlert[];
|
|
1777
|
+
/** Per-kind summary count. */
|
|
1778
|
+
counts: Record<CanaryKind, number>;
|
|
1779
|
+
/** Whether each enabled detector had enough observations to run. */
|
|
1780
|
+
evaluations: CanaryEvaluation[];
|
|
1781
|
+
}
|
|
1782
|
+
interface CanaryEvaluation {
|
|
1783
|
+
kind: CanaryKind;
|
|
1784
|
+
status: 'evaluated' | 'not_evaluated';
|
|
1785
|
+
observations: number;
|
|
1786
|
+
reason?: string;
|
|
1787
|
+
}
|
|
1788
|
+
interface CanaryOptions {
|
|
1789
|
+
/**
|
|
1790
|
+
* Silent-fallback detection.
|
|
1791
|
+
* - `constant`: confidence value treated as the fallback signal.
|
|
1792
|
+
* Default 0.30 (matches the soft-fail default in
|
|
1793
|
+
* `propose-review.ts`).
|
|
1794
|
+
* - `consecutiveThreshold`: trip the alert after this many
|
|
1795
|
+
* consecutive runs at `constant` (or `fallback === true`).
|
|
1796
|
+
* Default 3.
|
|
1732
1797
|
*/
|
|
1733
|
-
|
|
1798
|
+
silentFallback?: {
|
|
1799
|
+
constant?: number;
|
|
1800
|
+
consecutiveThreshold?: number;
|
|
1801
|
+
/** Floating-point tolerance when comparing against `constant`. */
|
|
1802
|
+
epsilon?: number;
|
|
1803
|
+
};
|
|
1734
1804
|
/**
|
|
1735
|
-
*
|
|
1736
|
-
*
|
|
1737
|
-
*
|
|
1738
|
-
*
|
|
1739
|
-
*
|
|
1740
|
-
*
|
|
1741
|
-
*
|
|
1742
|
-
*
|
|
1743
|
-
*
|
|
1744
|
-
* the provided seeded policy.
|
|
1805
|
+
* Calibration-drift detection.
|
|
1806
|
+
* - `historyWindow`: number of past runs (oldest-first) treated as
|
|
1807
|
+
* the historical baseline. Default 50.
|
|
1808
|
+
* - `recentWindow`: number of recent runs (newest-first) compared
|
|
1809
|
+
* against history. Default 20.
|
|
1810
|
+
* - `ksAlpha`: alpha for the KS statistic vs critical value.
|
|
1811
|
+
* Default 0.05.
|
|
1812
|
+
* - `minRecent`: minimum recent runs required to even attempt the
|
|
1813
|
+
* check. Default 10.
|
|
1745
1814
|
*/
|
|
1746
|
-
|
|
1815
|
+
calibrationDrift?: {
|
|
1816
|
+
historyWindow?: number;
|
|
1817
|
+
recentWindow?: number;
|
|
1818
|
+
ksAlpha?: number;
|
|
1819
|
+
minRecent?: number;
|
|
1820
|
+
};
|
|
1747
1821
|
/**
|
|
1748
|
-
*
|
|
1749
|
-
*
|
|
1750
|
-
*
|
|
1751
|
-
*
|
|
1752
|
-
* `
|
|
1753
|
-
*
|
|
1754
|
-
* `identity` declares what the ledger requires and a campaign cannot infer:
|
|
1755
|
-
* immutable revisions for the agent, proposer, and search implementations,
|
|
1756
|
-
* and the model the agent runs when a cell reports none.
|
|
1822
|
+
* Distribution-shift detection.
|
|
1823
|
+
* - `category`: function that maps a run to a categorical bucket.
|
|
1824
|
+
* Required to enable this canary; if omitted the chi-square check
|
|
1825
|
+
* is skipped entirely.
|
|
1826
|
+
* - `chiSquareAlpha`: alpha. Default 0.05.
|
|
1827
|
+
* - `historyWindow`, `recentWindow`, `minRecent`: like above.
|
|
1757
1828
|
*/
|
|
1758
|
-
|
|
1759
|
-
|
|
1760
|
-
|
|
1761
|
-
|
|
1762
|
-
|
|
1763
|
-
|
|
1764
|
-
|
|
1765
|
-
surfaceHash: string;
|
|
1766
|
-
surface: MutableSurface;
|
|
1767
|
-
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1768
|
-
}>;
|
|
1769
|
-
}>;
|
|
1770
|
-
/** Frozen snapshot of the exact starting surface measured by `baselineCampaign`. */
|
|
1771
|
-
baselineSurface: MutableSurface;
|
|
1772
|
-
winnerSurface: MutableSurface;
|
|
1773
|
-
winnerSurfaceHash: string;
|
|
1774
|
-
/** Proposer label for the promoted surface. Present when the winning
|
|
1775
|
-
* candidate came from a `ProposedCandidate` (a reflective proposer);
|
|
1776
|
-
* absent when the winner is the baseline or a bare-surface mutator. */
|
|
1777
|
-
winnerLabel?: string;
|
|
1778
|
-
/** Proposer rationale for the promoted surface — the "because Z" that
|
|
1779
|
-
* motivated the winning change. Survives to `SelfImproveResult` and the
|
|
1780
|
-
* emitted provenance record. Absent when the winner is the baseline. */
|
|
1781
|
-
winnerRationale?: string;
|
|
1782
|
-
baselineCampaign: CampaignResult<TArtifact, TScenario>;
|
|
1783
|
-
/** Run-wide spend, including agents, proposers, analysts, and judges. */
|
|
1784
|
-
cost: CostLedgerSummary;
|
|
1785
|
-
/** Bounded proof envelope over the canonical search ledger. Present only
|
|
1786
|
-
* when `searchLedger` was supplied. `complete` is false when the search was
|
|
1787
|
-
* interrupted or a candidate left a designed cell unscored. */
|
|
1788
|
-
searchHistory?: SearchHistoryReceipt;
|
|
1789
|
-
/** The GEPA Pareto frontier across every scored surface (baseline + all
|
|
1790
|
-
* generations) by per-scenario objective vector — the non-dominated set.
|
|
1791
|
-
* Each generation's `propose()` received the frontier-so-far as
|
|
1792
|
-
* `ctx.paretoParents`; this is the final frontier. A surface here that is
|
|
1793
|
-
* NOT the winner is uniquely best on some scenario the winner loses on. */
|
|
1794
|
-
paretoFrontier: ParetoParent[];
|
|
1829
|
+
distributionShift?: {
|
|
1830
|
+
category: (run: RunRecord) => string | null;
|
|
1831
|
+
chiSquareAlpha?: number;
|
|
1832
|
+
historyWindow?: number;
|
|
1833
|
+
recentWindow?: number;
|
|
1834
|
+
minRecent?: number;
|
|
1835
|
+
};
|
|
1795
1836
|
}
|
|
1796
1837
|
/**
|
|
1797
|
-
*
|
|
1838
|
+
* Run all configured canaries against a chronological run list.
|
|
1839
|
+
* Runs MUST be sorted oldest-to-newest by the caller — the order of
|
|
1840
|
+
* the input is used to define "recent" vs "historical" windows.
|
|
1798
1841
|
*/
|
|
1799
|
-
declare function
|
|
1842
|
+
declare function runCanaries(runs: RunRecord[], opts?: CanaryOptions): CanaryReport;
|
|
1800
1843
|
//#endregion
|
|
1801
|
-
//#region src/
|
|
1802
|
-
type
|
|
1803
|
-
|
|
1804
|
-
|
|
1805
|
-
|
|
1806
|
-
|
|
1807
|
-
|
|
1808
|
-
*
|
|
1809
|
-
*
|
|
1810
|
-
|
|
1811
|
-
|
|
1812
|
-
|
|
1813
|
-
|
|
1814
|
-
|
|
1815
|
-
|
|
1816
|
-
|
|
1817
|
-
|
|
1818
|
-
|
|
1819
|
-
|
|
1820
|
-
|
|
1821
|
-
|
|
1822
|
-
|
|
1823
|
-
|
|
1824
|
-
|
|
1825
|
-
|
|
1826
|
-
|
|
1827
|
-
|
|
1828
|
-
|
|
1829
|
-
|
|
1830
|
-
|
|
1831
|
-
* a `neutralizationGate` reject a win whose lift survives blanking the content
|
|
1832
|
-
* (decorative — driven by footprint, not content). Costs one extra holdout
|
|
1833
|
-
* campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
|
|
1834
|
-
neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
|
|
1835
|
-
};
|
|
1836
|
-
interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
|
|
1837
|
-
baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1838
|
-
winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1839
|
-
neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
|
|
1840
|
-
neutralizedSurface?: MutableSurface;
|
|
1841
|
-
gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
|
|
1842
|
-
/** Present iff the loop ran with `holdout: 'deferred'`. When set,
|
|
1843
|
-
* `baselineOnHoldout`/`winnerOnHoldout` are the shared EMPTY campaign (zero
|
|
1844
|
-
* cells dispatched) and the gate verdict is the forced `'hold'`. */
|
|
1845
|
-
holdout?: 'deferred';
|
|
1846
|
-
/** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
|
|
1847
|
-
* when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
|
|
1848
|
-
* always present on the result + in the emitted provenance record. Empty
|
|
1849
|
-
* string when winner == baseline (no change to diff). */
|
|
1850
|
-
promotedDiff: string;
|
|
1851
|
-
prResult?: ReturnType<typeof openAutoPr>;
|
|
1844
|
+
//#region src/red-team.d.ts
|
|
1845
|
+
type RedTeamCategory = 'prompt_injection_direct' | 'prompt_injection_indirect' | 'jailbreak_persona' | 'jailbreak_dan' | 'pii_leak' | 'permission_escalation' | 'data_exfil' | 'policy_override';
|
|
1846
|
+
interface RedTeamPayload {
|
|
1847
|
+
category: RedTeamCategory;
|
|
1848
|
+
/** The hostile input the agent sees. */
|
|
1849
|
+
input: string;
|
|
1850
|
+
/**
|
|
1851
|
+
* What a safe agent should do: "refuse" means explicit refusal,
|
|
1852
|
+
* "ignore" means process benignly without executing the attack.
|
|
1853
|
+
*/
|
|
1854
|
+
expected: 'refuse' | 'ignore';
|
|
1855
|
+
/** Substring(s) that MUST NOT appear in output. */
|
|
1856
|
+
forbiddenStrings?: string[];
|
|
1857
|
+
/** Tool names the agent MUST NOT call for this scenario. */
|
|
1858
|
+
forbiddenTools?: string[];
|
|
1859
|
+
}
|
|
1860
|
+
interface RedTeamCase extends DatasetScenario {
|
|
1861
|
+
payload: RedTeamPayload;
|
|
1862
|
+
}
|
|
1863
|
+
interface RedTeamFinding {
|
|
1864
|
+
scenarioId: string;
|
|
1865
|
+
category: RedTeamCategory;
|
|
1866
|
+
passed: boolean;
|
|
1867
|
+
reason: string;
|
|
1868
|
+
evidence?: string;
|
|
1869
|
+
}
|
|
1870
|
+
interface RedTeamReport {
|
|
1871
|
+
findings: RedTeamFinding[];
|
|
1872
|
+
passRateByCategory: Record<RedTeamCategory, number>;
|
|
1873
|
+
overallPassRate: number;
|
|
1852
1874
|
}
|
|
1875
|
+
/** Ship a canonical, small corpus. Consumers extend via `extendCorpus`. */
|
|
1876
|
+
declare const DEFAULT_RED_TEAM_CORPUS: RedTeamCase[];
|
|
1877
|
+
declare function redTeamDataset(extraCases?: RedTeamCase[]): Dataset;
|
|
1853
1878
|
/**
|
|
1854
|
-
*
|
|
1879
|
+
* Score a single agent response against the red-team case.
|
|
1880
|
+
* `toolCalls` should be the names of tools the agent invoked during the run.
|
|
1855
1881
|
*/
|
|
1856
|
-
declare function
|
|
1882
|
+
declare function scoreRedTeamOutput(output: string, toolCalls: string[], rtCase: RedTeamCase): RedTeamFinding;
|
|
1883
|
+
/** Aggregate red-team findings into per-category pass rates. */
|
|
1884
|
+
declare function redTeamReport(findings: RedTeamFinding[]): RedTeamReport;
|
|
1857
1885
|
//#endregion
|
|
1858
1886
|
//#region src/campaign/provenance.d.ts
|
|
1859
1887
|
interface LoopProvenanceCandidate {
|
|
@@ -1870,6 +1898,10 @@ interface LoopProvenanceCandidate {
|
|
|
1870
1898
|
/** Proposer rationale — the "because Z". When the proposer returned a bare
|
|
1871
1899
|
* surface (blind mutator) this is absent. */
|
|
1872
1900
|
rationale?: string;
|
|
1901
|
+
/** Proposer-supplied typed attribution, carried unchanged from
|
|
1902
|
+
* `GenerationCandidate.attribution`. Opaque here; the producer's schema tag
|
|
1903
|
+
* governs interpretation. */
|
|
1904
|
+
attribution?: Readonly<Record<string, unknown>>;
|
|
1873
1905
|
/** Exact complete incumbent this candidate mutated. */
|
|
1874
1906
|
parentSurfaceHash: string;
|
|
1875
1907
|
/** Search-split composite of the exact parent. */
|
|
@@ -2077,33 +2109,5 @@ interface EmitLoopProvenanceArgs<TArtifact, TScenario extends Scenario> extends
|
|
|
2077
2109
|
*/
|
|
2078
2110
|
declare function emitLoopProvenance<TArtifact, TScenario extends Scenario>(args: EmitLoopProvenanceArgs<TArtifact, TScenario>): Promise<EmitLoopProvenanceResult>;
|
|
2079
2111
|
//#endregion
|
|
2080
|
-
|
|
2081
|
-
|
|
2082
|
-
/**
|
|
2083
|
-
* Treat full-duration timeouts ("timeout after 180000ms") as transient.
|
|
2084
|
-
* Enable on saturated shared infrastructure where queue starvation eats
|
|
2085
|
-
* the clock; leave off when the agent had the resources and simply failed.
|
|
2086
|
-
* Default false.
|
|
2087
|
-
*/
|
|
2088
|
-
readonly retryFullDurationTimeouts?: boolean;
|
|
2089
|
-
/** Additional caller-specific transient patterns. */
|
|
2090
|
-
readonly extraPatterns?: readonly RegExp[];
|
|
2091
|
-
}
|
|
2092
|
-
/**
|
|
2093
|
-
* True when the error text describes an infrastructure hiccup that should be
|
|
2094
|
-
* retried rather than scored. Empty/undefined input is not transient.
|
|
2095
|
-
*/
|
|
2096
|
-
declare function isTransientTransportFailure(message: string | null | undefined, opts?: TransientFailureOptions): boolean;
|
|
2097
|
-
/**
|
|
2098
|
-
* Ready-made `cellRetry.retryable` predicate: true for a dispatch-stage
|
|
2099
|
-
* failure whose error message `isTransientTransportFailure` classifies as an
|
|
2100
|
-
* infrastructure hiccup. A judge-stage failure is never retried here — the
|
|
2101
|
-
* dispatch already produced an artifact, so re-dispatching would score a
|
|
2102
|
-
* different sample. A per-cell dispatch deadline ("dispatch exceeded <N>ms")
|
|
2103
|
-
* is not transient by default; opt in via `extraPatterns` or
|
|
2104
|
-
* `retryFullDurationTimeouts` when queue starvation eats the clock.
|
|
2105
|
-
*/
|
|
2106
|
-
declare function transientDispatchFailure(opts?: TransientFailureOptions): (failure: CampaignCellFailureReceipt['failure']) => boolean;
|
|
2107
|
-
//#endregion
|
|
2108
|
-
export { CanaryKind as $, ServedModelVerdict as $n, SearchOperationKind as $t, runEval as A, CampaignCellRetryPolicy as An, verifySearchHistoryReceipt as At, SearchRecorderOptions as B, inMemoryCampaignStorage as Bn, SearchCandidateSlotClosedEvent as Bt, RunImprovementLoopResult as C, ExternalOptimizerSubmittedCandidate as Cn, SearchHistoryPolicy as Ct, RunOptimizationResult as D, readCachedCell as Dn, assertSearchHistoryMatchesReplay as Dt, RunOptimizationOptions as E, CacheRead as En, assertCompleteSearchHistory as Et, MeasuredSearchCandidate as F, PlanCampaignRunOptions as Fn, SearchAttemptAccounting as Ft, RedTeamCategory as G, LlmJudgeOptions as Gn, SearchLedger as Gt, recordCandidatePopulationSearch as H, OpenAutoPrResult as Hn, SearchCompletedEvent as Ht, ProposedSearchCandidate as I, planCampaignRun as In, SearchCandidateDecidedEvent as It, redTeamDataset as J, AssertServedModelOptions as Jn, SearchLedgerEvent as Jt, RedTeamFinding as K, llmJudge as Kn, SearchLedgerAppendResult as Kt, SearchExecutionIdentity as L, CampaignStorage as Ln, SearchCandidateLineage as Lt, ParentSelectionContext as M, runCampaign as Mn, OpenSearchLedgerOptions as Mt, ParentSelector as N, CampaignRunPlan as Nn, SearchAccountingAudit as Nt, runOptimization as O, cellCachePath as On, createSearchHistoryReceipt as Ot, crowdedFrontierParent as P, CampaignRunPlanCell as Pn, SearchArtifactRef as Pt, CanaryEvaluation as Q, ServedModelPolicy as Qn, SearchModelIdentity as Qt, SearchLedgerBinding as R, createRunCostLedger as Rn, SearchCandidateRegisteredEvent as Rt, RunImprovementLoopOptions as S, ExternalOptimizerObservationSummary as Sn, SearchHistoryCoverageRow as St, PremeasuredOptimizationBaseline as T, CacheIssueReason as Tn, SearchHistoryRequiredError as Tt, DEFAULT_RED_TEAM_CORPUS as U, openAutoPr as Un, SearchCostAccounting as Ut, SearchRunIdentity as V, OpenAutoPrOptions as Vn, SearchCandidateSurface as Vt, RedTeamCase as W, LlmJudgeDimension as Wn, SearchFailureReason as Wt, scoreRedTeamOutput as X, ServedCrossFamilyError as Xn, SearchLedgerReplay as Xt, redTeamReport as Y, ModelSubstitutionError as Yn, SearchLedgerHash as Yt, CanaryAlert as Z, ServedModelCheck as Zn, SearchLedgerTrustedHeadMode as Zt, loopProvenanceArgsFromResult as _, GepaCandidatePopulationSummary as _n, costFromLedgerSummary as _t, EmitLoopProvenanceArgs as a, SearchPlannedTask as an, AssertCrossFamilyOptions as ar, OptimizationMethod as at, provenanceSpansPath as b, ExternalOptimizerExecutionSummary as bn, SearchHistoryAuditSummary as bt, LoopProvenanceBackend as c, SearchSurfaceEvidence as cn, assertCrossFamily as cr, OptimizationMethodPairwise as ct, LoopProvenanceOptimizationMethod as d, SearchTaskOutcome as dn, OptimizationMethodRunOptions as dt, SearchOperationRecordedEvent as en, assertCrossFamilyServed as er, CanaryOptions as et, LoopProvenanceRecord as f, SearchTokenAccounting as fn, OptimizationMethodScore as ft, emitLoopProvenance as g, GepaCandidatePopulationCandidate as gn, compareOptimizationMethods as gt, canonicalDigest as h, GepaCandidatePopulationArtifact as hn, combineComparisonCosts as ht, BuildLoopProvenanceArgs as i, SearchPlannedOperation as in, servedModelAcceptable as ir, ComparisonCost as it, CrowdedFrontierParentOptions as j, RunCampaignOptions as jn, FileSearchLedger as jt, RunEvalOptions as k, CampaignCellFailureReceipt as kn, searchHistoryCoverageRow as kt, LoopProvenanceCandidate as l, SearchSurfaceKind as ln, judgeFamily as lr, OptimizationMethodProvenance as lt, campaignMeasurementDigest as m, validateSearchLedgerEvent as mn, OptimizationTokenUsage as mt, isTransientTransportFailure as n, SearchPlanExtendedEvent as nn, assertServedModels as nr, runCanaries as nt, EmitLoopProvenanceResult as o, SearchSourceRef as on, CrossFamilyError as or, OptimizationMethodComparison as ot, buildLoopProvenanceRecord as p, openSearchLedger as pn, OptimizationPackageSource as pt, RedTeamReport as q, AssertCrossFamilyServedOptions as qn, SearchLedgerEntry as qt, transientDispatchFailure as r, SearchPlannedEvent as rn, checkServedModel as rr, CompareOptimizationMethodsOptions as rt, LoopProvenanceArgsFromResult as s, SearchSurfaceEffect as sn, JudgeFamily as sr, OptimizationMethodInput as st, TransientFailureOptions as t, SearchPlan as tn, assertServedModel as tr, CanaryReport as tt, LoopProvenanceEvidence as u, SearchTaskAttemptedEvent as un, OptimizationMethodResult as ut, loopProvenanceSpans as v, GepaCandidateSelectionScore as vn, optimizationTokenUsageFromSummary as vt, runImprovementLoop as w, readExternalOptimizerObservationArtifact as wn, SearchHistoryReceipt as wt, verifyLoopProvenanceRecord as x, ExternalOptimizerObservationArtifact as xn, SearchHistoryCoverage as xt, provenanceRecordPath as y, readGepaCandidatePopulationArtifact as yn, CreateSearchHistoryReceiptInput as yt, SearchRecorder as z, fsCampaignStorage as zn, SearchCandidateSlot as zt };
|
|
2109
|
-
//# sourceMappingURL=transient-failure-DKF5Mofa.d.ts.map
|
|
2112
|
+
export { readExternalOptimizerObservationArtifact as $, ServedModelVerdict as $n, SearchLedgerAppendResult as $t, CanaryOptions as A, runEval as An, SearchHistoryPolicy as At, OptimizationMethodResult as B, CacheRead as Bn, SearchAccountingAudit as Bt, RedTeamReport as C, ParentSelectionContext as Cn, GepaCandidatePopulationSummary as Ct, CanaryAlert as D, OpenAutoPrResult as Dn, SearchHistoryAuditSummary as Dt, scoreRedTeamOutput as E, OpenAutoPrOptions as En, CreateSearchHistoryReceiptInput as Et, OptimizationMethod as F, CampaignRunPlan as Fn, createSearchHistoryReceipt as Ft, combineComparisonCosts as G, fsCampaignStorage as Gn, SearchCandidateRegisteredEvent as Gt, OptimizationMethodScore as H, cellCachePath as Hn, SearchAttemptAccounting as Ht, OptimizationMethodComparison as I, CampaignRunPlanCell as In, searchHistoryCoverageRow as It, optimizationTokenUsageFromSummary as J, AssertServedModelOptions as Jn, SearchCandidateSurface as Jt, compareOptimizationMethods as K, inMemoryCampaignStorage as Kn, SearchCandidateSlot as Kt, OptimizationMethodInput as L, PlanCampaignRunOptions as Ln, verifySearchHistoryReceipt as Lt, runCanaries as M, CampaignCellRetryPolicy as Mn, SearchHistoryRequiredError as Mt, CompareOptimizationMethodsOptions as N, RunCampaignOptions as Nn, assertCompleteSearchHistory as Nt, CanaryEvaluation as O, openAutoPr as On, SearchHistoryCoverage as Ot, ComparisonCost as P, runCampaign as Pn, assertSearchHistoryMatchesReplay as Pt, ExternalOptimizerSubmittedCandidate as Q, ServedModelPolicy as Qn, SearchLedger as Qt, OptimizationMethodPairwise as R, planCampaignRun as Rn, FileSearchLedger as Rt, RedTeamFinding as S, CrowdedFrontierParentOptions as Sn, GepaCandidatePopulationCandidate as St, redTeamReport as T, crowdedFrontierParent as Tn, readGepaCandidatePopulationArtifact as Tt, OptimizationPackageSource as U, CampaignStorage as Un, SearchCandidateDecidedEvent as Ut, OptimizationMethodRunOptions as V, readCachedCell as Vn, SearchArtifactRef as Vt, OptimizationTokenUsage as W, createRunCostLedger as Wn, SearchCandidateLineage as Wt, ExternalOptimizerObservationArtifact as X, ServedCrossFamilyError as Xn, SearchCostAccounting as Xt, ExternalOptimizerExecutionSummary as Y, ModelSubstitutionError as Yn, SearchCompletedEvent as Yt, ExternalOptimizerObservationSummary as Z, ServedModelCheck as Zn, SearchFailureReason as Zt, provenanceSpansPath as _, SearchTaskAttemptedEvent as _n, SearchRecorder as _t, LoopProvenanceBackend as a, SearchModelIdentity as an, AssertCrossFamilyOptions as ar, transientDispatchFailure as at, RedTeamCase as b, openSearchLedger as bn, recordCandidatePopulationSearch as bt, LoopProvenanceOptimizationMethod as c, SearchPlan as cn, assertCrossFamily as cr, runImprovementLoop as ct, campaignMeasurementDigest as d, SearchPlannedOperation as dn, RunOptimizationResult as dt, SearchLedgerEntry as en, assertCrossFamilyServed as er, LlmJudgeDimension as et, canonicalDigest as f, SearchPlannedTask as fn, runOptimization as ft, provenanceRecordPath as g, SearchSurfaceKind as gn, SearchLedgerBinding as gt, loopProvenanceSpans as h, SearchSurfaceEvidence as hn, SearchExecutionIdentity as ht, LoopProvenanceArgsFromResult as i, SearchLedgerTrustedHeadMode as in, servedModelAcceptable as ir, isTransientTransportFailure as it, CanaryReport as j, CampaignCellFailureReceipt as jn, SearchHistoryReceipt as jt, CanaryKind as k, RunEvalOptions as kn, SearchHistoryCoverageRow as kt, LoopProvenanceRecord as l, SearchPlanExtendedEvent as ln, judgeFamily as lr, PremeasuredOptimizationBaseline as lt, loopProvenanceArgsFromResult as m, SearchSurfaceEffect as mn, ProposedSearchCandidate as mt, EmitLoopProvenanceArgs as n, SearchLedgerHash as nn, assertServedModels as nr, llmJudge as nt, LoopProvenanceCandidate as o, SearchOperationKind as on, CrossFamilyError as or, RunImprovementLoopOptions as ot, emitLoopProvenance as p, SearchSourceRef as pn, MeasuredSearchCandidate as pt, costFromLedgerSummary as q, AssertCrossFamilyServedOptions as qn, SearchCandidateSlotClosedEvent as qt, EmitLoopProvenanceResult as r, SearchLedgerReplay as rn, checkServedModel as rr, TransientFailureOptions as rt, LoopProvenanceEvidence as s, SearchOperationRecordedEvent as sn, JudgeFamily as sr, RunImprovementLoopResult as st, BuildLoopProvenanceArgs as t, SearchLedgerEvent as tn, assertServedModel as tr, LlmJudgeOptions as tt, buildLoopProvenanceRecord as u, SearchPlannedEvent as un, RunOptimizationOptions as ut, verifyLoopProvenanceRecord as v, SearchTaskOutcome as vn, SearchRecorderOptions as vt, redTeamDataset as w, ParentSelector as wn, GepaCandidateSelectionScore as wt, RedTeamCategory as x, validateSearchLedgerEvent as xn, GepaCandidatePopulationArtifact as xt, DEFAULT_RED_TEAM_CORPUS as y, SearchTokenAccounting as yn, SearchRunIdentity as yt, OptimizationMethodProvenance as z, CacheIssueReason as zn, OpenSearchLedgerOptions as zt };
|
|
2113
|
+
//# sourceMappingURL=provenance-CafMdZKM.d.ts.map
|