@tangle-network/agent-eval 0.180.0 → 0.181.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +60 -0
- package/README.md +119 -159
- package/dist/adapters/http.d.ts +2 -2
- package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
- package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
- package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
- package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
- package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
- package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +3 -3
- package/dist/ast-CP9ae9B0.js +557 -0
- package/dist/ast-CP9ae9B0.js.map +1 -0
- package/dist/ast-hI-vjW6J.d.ts +457 -0
- package/dist/ast-hI-vjW6J.d.ts.map +1 -0
- package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
- package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
- package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
- package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
- package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
- package/dist/cli.js +4 -7
- package/dist/cli.js.map +1 -1
- package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
- package/dist/client-CXE-U1SA.js.map +1 -0
- package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
- package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +10 -9
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
- package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
- package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
- package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
- package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
- package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
- package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
- package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
- package/dist/engine-DS1cysJy.d.ts.map +1 -0
- package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
- package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
- package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
- package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +27 -477
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +95 -559
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
- package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
- package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
- package/dist/index-Bp_6sj3x.d.ts.map +1 -0
- package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
- package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
- package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
- package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
- package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
- package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
- package/dist/index.d.ts +28 -28
- package/dist/index.js +24 -15
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
- package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
- package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
- package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
- package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
- package/dist/journal-Cs9f7385.js.map +1 -0
- package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
- package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
- package/dist/llm-judge-DEFZeSiu.js.map +1 -0
- package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
- package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +138 -7
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +245 -97
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
- package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/outcome-store-BXlkwMPR.js +131 -0
- package/dist/outcome-store-BXlkwMPR.js.map +1 -0
- package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
- package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
- package/dist/pipelines/index.js +1 -1
- package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
- package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
- package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
- package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
- package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
- package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
- package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
- package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
- package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +3 -3
- package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
- package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
- package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
- package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
- package/dist/rl.d.ts +53 -99
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +182 -169
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
- package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
- package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
- package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
- package/dist/run-record-Br-Yzt_k.js +464 -0
- package/dist/run-record-Br-Yzt_k.js.map +1 -0
- package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
- package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
- package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
- package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
- package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
- package/dist/sequential-DAsyV2T9.js.map +1 -0
- package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
- package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
- package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
- package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
- package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
- package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
- package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
- package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
- package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
- package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
- package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
- package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
- package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
- package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
- package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
- package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
- package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
- package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
- package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
- package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
- package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
- package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/docs/adapters-observability.md +14 -0
- package/docs/campaign-proposers.md +86 -128
- package/docs/charter.md +108 -112
- package/docs/concepts.md +157 -69
- package/docs/design/mlbenchmarks-book-review.md +440 -0
- package/docs/design/mlbenchmarks-review/observations.json +713 -0
- package/docs/design/mlbenchmarks-review/probes.mts +476 -0
- package/docs/design/mlbenchmarks-review/sources.json +200 -0
- package/docs/design/self-improvement-evidence-audit.md +263 -0
- package/docs/design.md +2 -1
- package/docs/eval-surface-map.md +95 -42
- package/docs/evaluation-integrity.md +220 -0
- package/docs/experiment.md +111 -55
- package/docs/feature-guide.md +5 -6
- package/docs/hosted-ingest-spec.md +4 -11
- package/docs/insight-report.md +187 -455
- package/docs/outcome-validity.md +182 -0
- package/docs/product-eval-adoption.md +1 -2
- package/docs/research-report-methodology.md +7 -7
- package/docs/search-history-receipts.md +8 -0
- package/docs/statistical-evidence.md +129 -0
- package/docs/verdicts.md +76 -49
- package/package.json +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
- package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
- package/dist/client-BlLY6o2w.js.map +0 -1
- package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
- package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
- package/dist/engine-CX8ReXkn.d.ts.map +0 -1
- package/dist/index-BxWvILU8.d.ts.map +0 -1
- package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
- package/dist/ledger-core-Cs9f7385.js.map +0 -1
- package/dist/llm-judge-v80Kmu9g.js.map +0 -1
- package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
- package/dist/outcome-store-ChBKlTd_.js +0 -75
- package/dist/outcome-store-ChBKlTd_.js.map +0 -1
- package/dist/promotion-policy-DWOm70gx.js.map +0 -1
- package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
- package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
- package/dist/run-record-CR63CpHK.js +0 -216
- package/dist/run-record-CR63CpHK.js.map +0 -1
- package/dist/sequential-B5gXgcyp.js.map +0 -1
- package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
- package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types-
|
|
1
|
+
{"version":3,"file":"types-CS0qk_Yp.d.ts","names":[],"sources":["../src/campaign/types.ts"],"mappings":";;;;;;;;UAiCiB;EACf;EACA;EACA;;;;;EAKA;;;UAIe,iCAAiC,KAAK;EACrD;;;;;UAMe;EACf;;EAEA;EACA;EACA;EACA;EACA,QAAQ;EACR,OAAO;EACP,WAAW;EACX,MAAM;;EAEN;;EAEA;;;;;;;;EAQA;;;;KAKU,WAAW,kBAAkB,UAAU,cACjD,UAAU,WACV,KAAK,oBACF,QAAQ;;;;UAOI,cAAc,WAAW;EACxC;EACA;EACA;;EAEA;;;EAGA,sBAAsB,UAAU,WAAW,sBAAsB,UAAU,cAAc;;UAK1E;;EAEf;;EAEA;;;;;;;;;UAUe,YAAY,WAAW,kBAAkB,WAAW;EACnE;EACA,YAAY;;;;EAIZ;;;EAGA,MAAM;IACJ,UAAU;IACV,UAAU;IACV,QAAQ;;IAER,aAAa;IACb;IACA,WAAW;MACT,aAAa,QAAQ;EACzB,aAAa,UAAU;;;;;;;;;;;UAYR;EACf,YAAY;EACZ;EACA;;EAEA,UAAU;;;;;;;EAOV;;;;EAIA,eAAe,eAAe;IAAQ;IAAe;;;;;EAIrD;;;EAGA;;EAEA;;EAEA,WAAW,eAAe;;;;;;;UAUX;WACN;;;WAGA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;;WAGA;aACE;aACA;aACA;;;WAGF;;;UAIM;WACN;WACA,YAAY,SAAS;;;;;;;;;;;KAYpB,0BAA0B,mBAAmB;;;;;;;UAQxC;EACf,SAAS;;EAET;;;;EAIA;;;;;;EAMA,cAAc,SAAS;;;;iBAKT,oBACd,OAAO,iBAAiB,oBACvB,SAAS;;;;;;;;;UAkBK;EACf,SAAS;EACT;;;EAGA,YAAY;;;EAGZ;;EAEA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EACA;;;;;UAMe;;;EAGf;;EAEA;EACA;EACA;EACA,YAAY;;;;;EAKZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;EAC1E;IACE;IACA;;;;;UAMa,eAAe,YAAY;;;WAGjC,gBAAgB;WAChB,SAAS,cAAc;WACvB,UAAU,cAAc;;WAExB;WACA;WACA,QAAQ;;WAER,QAAQ;;;WAGR,kBAAkB;;;WAGlB,mBAAmB;;;;WAInB,gBAAgB;;;;WAIhB;;;;;;;WAOA,gBAAgB,cAAc;;WAE9B,aAAa;WACb;;;;;;;;;;;;;;UAeM,gBAAgB,YAAY;EAC3C;;;;;EAKA,QAAQ,KAAK,eAAe,aAAa,QAAQ,MAAM,iBAAiB;;;EAGxE,QAAQ;IAAQ,SAAS,cAAc;;IAAwB;IAAe;;;UAG/D;EACf;EACA;EACA,mBAAmB,qBAAqB;;UAGzB,wBAAwB;EACvC,UAAU;;;KAMA;;KAGA;UAEK;EACf;EACA,QAAQ;EACR;;UAGe,YAAY,WAAW,kBAAkB;EACxD,oBAAoB,YAAY;EAChC,oBAAoB,YAAY;;EAEhC,aAAa,YAAY,eAAe;;;;;EAKxC,sBAAsB,YAAY,eAAe;;;;;;;EAOjD,yBAAyB,YAAY,eAAe;;;EAGpD,uBAAuB,YAAY;EACnC,WAAW;EACX;IAAQ;IAAmB;;;EAE3B,aAAa;EACb;EACA,QAAQ;;UAGO;EACf,UAAU;EACV;EACA,mBAAmB;EACnB;;;UAIe,KAAK,qBAAqB,kBAAkB,WAAW;EACtE;EACA,OAAO,KAAK,YAAY,WAAW,aAAa,QAAQ;;;;UAOzC;EACf,KAAK,cAAc,aAAa,0BAA0B;EAC1D,SAAS;;UAGM;EACf,IAAI,aAAa;EACjB,aAAa,aAAa;;;;UAKX;EACf,MAAM,cAAc,kBAAkB,aAAa;EACnD,UAAU,cAAc,iBAAiB;;;;;;KAO/B,qBAAqB;;;;;UAMhB;;EAEf,YAAY,GACV,OAAO,KAAK,iBAAiB;IAC3B,UAAU;MAEX,QAAQ,eAAe;;;;;KAQhB;KAOA;;;;;;;;;;;;;;;KAgBA;;iBASI,eAAe,OAAO;;;;UAOrB,qBAAqB,kBAAkB,WAAW,UAAU;EAC3E,UAAU;EACV,UAAU;EACV,aAAa,eAAe;EAC5B,QAAQ;EACR;EACA;EACA,iBAAiB;;;;;EAKjB,aAAa;;EAEb;;UAGe,sBAAsB,kBAAkB,WAAW,UAAU,6BACpE,qBAAqB,WAAW;;EAExC;;;EAGA;;UAGe;EACf;;EAEA;;;;EAIA;EACA;IACE;IACA,SAAS,wBAAwB;IACjC;IACA;;;;;IAKA,WAAW;;;UAIE;EACf,QAAQ,OAAO,uBAAuB;EACtC,OAAO,MAAM,4BAA4B,QAAQ;EACjD,QAAQ;IACN;IACA;IACA,UAAU;;;IAGV,SAAS,OAAO;;;UAMH,mBAAmB;;;EAGlC;EACA;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa,eAAe;;;EAG5B;;EAEA,gBAAgB;;EAEhB;;;EAGA,YAAY;;;EAGZ;;;EAGA;EACA;EACA;EACA;;;;EAIA;;EAEA;;EAEA;EACA;;UAGe;EACf;EACA;EACA;EACA;;;;;;EAMA,cAAc;;UAGC;EACf;EACA;EACA;;;EAGA,cAAc;;UAGC;EACf;EACA,YAAY;EACZ;;;;;;UAOe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;;EAIA;;;;EAIA;IACE;IACA;IACA,iBAAiB;MAAQ;MAAgB;;;;;EAI3C,YAAY;;;;;;;;;;EAUZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;;;EAG1E;;;;EAIA;;;;EAIA,cAAc,SAAS;;UAGR;EACf,SAAS,eAAe;EACxB,YAAY,eAAe;;EAE3B,MAAM;;EAEN;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe,eAAe,qBAAqB,kBAAkB,WAAW;;EAEhF;;EAEA;EACA;;EAEA;EACA;EACA;EACA;EACA,OAAO,MAAM,mBAAmB;EAChC,YAAY;EACZ;IACE,aAAa;IACb;;EAEF,OAAO;EACP;EACA;EACA,iBAAiB;;;EAGjB,WAAW,MAAM,2BAA2B,KAAK"}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { c as CostLedgerHandle } from "./cost-ledger-DbQdN3nO.js";
|
|
2
|
-
import { a as RunRecord, n as RunCostProvenance, u as RunTokenUsage } from "./run-record-
|
|
3
|
-
import {
|
|
2
|
+
import { a as RunRecord, n as RunCostProvenance, u as RunTokenUsage } from "./run-record-BiTWauyO.js";
|
|
3
|
+
import { c as JudgeInput, x as ChatClient } from "./types-CBbLtr2J.js";
|
|
4
4
|
import { RE2JS } from "re2js";
|
|
5
5
|
//#region src/trace-analyst/types.d.ts
|
|
6
6
|
/**
|
|
@@ -635,4 +635,4 @@ type AnalystRunEvent = {
|
|
|
635
635
|
};
|
|
636
636
|
//#endregion
|
|
637
637
|
export { DatasetOverview as A, TraceAnalystSpanKind as B, ReadSpanSourceInput as C, TraceAnalysisStore as D, TRACE_ANALYSIS_LIMITS as E, SpanMatchRecord as F, ViewTraceResult as G, TraceAnalystTraceSummary as H, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX as I, TraceAnalystByteBudgets as L, QueryTracesPage as M, SearchSpanResult as N, TraceAnalysisStoreContext as O, SearchTraceResult as P, TraceAnalystFilters as R, BoundedTraceAnalysisStoreOptions as S, SpanSourceReader as T, ViewSpansResult as U, TraceAnalystSpanStatus as V, ViewTraceOversized as W, ProposalFinding as _, AnalystInputKind as a, makeFinding as b, AnalystRunInputs as c, AnalystSeverity as d, AnalystUsageReceipt as f, ExecutionProbeRequest as g, ExecutionProbeOutcome as h, AnalystFinding as i, ErrorCluster as j, DEFAULT_TRACE_ANALYST_BUDGETS as k, AnalystRunResult as l, ExecutionProbe as m, AnalystContext as n, AnalystRequirements as o, EvidenceRef as p, AnalystCost as r, AnalystRunEvent as s, Analyst as t, AnalystRunSummary as u, ProposalFindingOrigin as v, ReadSpanSourceResult as w, makeProposalFinding as x, computeFindingId as y, TraceAnalystSpan as z };
|
|
638
|
-
//# sourceMappingURL=types-
|
|
638
|
+
//# sourceMappingURL=types-D7gEdPoQ.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types-
|
|
1
|
+
{"version":3,"file":"types-D7gEdPoQ.d.ts","names":[],"sources":["../src/trace-analyst/types.ts","../src/trace-analyst/store-contract.ts","../src/analyst/types.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;KAgBY;KAUA;;;;UAKK;EACf;EACA;EACA;EACA;EACA,MAAM;EACN;EACA;EACA;EACA,QAAQ;EACR;EACA;EACA;EACA;EACA;;;EAGA,YAAY;;UAGG;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;;;;;UAQe;;EAEf;;EAEA;;EAEA;;EAEA;EACA;EACA;;EAEA;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;;EAEA;EACA;IAAU;IAAqB;;;;;;EAK/B,gBAAgB;EAChB;IAAc;IAAkB;;;UAGjB;EACf,QAAQ;EACR;EACA;;;;;UAMe;EACf;EACA,QAAQ;EACR,YAAY;;UAGG;EACf;;EAEA,gBAAgB;;EAEhB;EACA;;UAGe;EACf;EACA,OAAO;;EAEP;;EAEA;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA;EACA,WAAW;;;EAGX;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA,MAAM;EACN;;UAGe;EACf;EACA;EACA,MAAM;EACN;;;UAIe;;;EAGf;;;EAGA;;;EAGA;;;EAGA;;cAGW,+BAA+B;;;cAS/B;;;cC7MA;WACX;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;UAGe;EACf,SAAS;;;UAIM;EACf;EACA;EACA;EACA;EACA;EACA;;KAGU;EAEN;EACA;EACA;EACA;EACA;EACA;;EAGA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;IACE;IACA;IACA;IACA;IACA;;;;KAKI,oBACV,OAAO,qBACP,UAAU,8BACP,QAAQ;;;;;;;;UASI;;EAEf,iBAAiB;EAEjB,SAAS,kBAAkB,UAAU,4BAA4B;EAEjE,SACE;IAAS;IAAkB;KAC3B,UAAU,4BACT;EAEH,YACE,UAAU,qBACV,UAAU,4BACT,QAAQ;EAEX,YACE;IAAS,UAAU;IAAqB;IAAe;KACvD,UAAU,4BACT,QAAQ;EAEX,YAAY,UAAU,qBAAqB,UAAU,4BAA4B;EAEjF,UACE;IACE;IACA;KAEF,UAAU,4BACT,QAAQ;EAEX,UACE;IACE;IACA;IACA;KAEF,UAAU,4BACT,QAAQ;EAEX,YACE;IACE;IACA;IACA;KAEF,UAAU,4BACT,QAAQ;EAEX,WACE;IACE;IACA;IACA;IACA;KAEF,UAAU,4BACT,QAAQ;;UAGI;EACf,UAAU,QAAQ;;;;;;;;UC9GH;EACf;;;;;;;EAOA;EACA;EACA;EACA,UAAU;;;;;;;EAOV;EACA;EACA;EACA,eAAe;EACf;EACA;;EAEA;;;;;;EAMA;;;;EAIA;;EAEA,WAAW;;;;cAKA;KAED,0BAA0B;;KAG1B;;KAGA,kBAAkB;WACnB,iBAAiB;;UAGX;;;;;;;EAOf;EACA;EACA;;;;;;;;KAWU;UAOK;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf;;EAEA;;;;;;UAOe;EACf,aAAa;EACb;EACA,YAAY;EACZ,aAAa;;EAEb,SAAS;;UAGM;EACf;;EAEA;;EAEA;;EAEA;;EAEA,aAAa;;EAEb;;;;;;;EAOA,OAAO;;;;;;;;;;EAUP,gBAAgB,cAAc;;;;;;;EAO9B,mBAAmB,cAAc;;;;;EAKjC,eAAe,SAAS;;EAExB,OAAO;;EAEP,OAAO,aAAa,SAAS;;EAE7B,SAAS;;;;;;;;EAQT,QAAQ;;;UAMO;EACf;;EAEA;;EAEA;;EAEA;EACA,SAAS;;;;;;;KAQC;EAEN;EACA;EACA;EACA;EACA;;EAEA;;EAEA;EAAkB;IAAS;IAAe;;;;;;;;UAO/B;;WAEN;EACT,QAAQ,SAAS,wBAAwB,QAAQ;;;;;;;;UASlC,QAAQ;;WAEd;;WAEA;WACA,WAAW;WACX,MAAM;WACN,WAAW;;WAEX;EACT,QAAQ,OAAO,QAAQ,KAAK,iBAAiB,QAAQ;;;UAItC;;EAEf;;EAEA,QAAQ;;EAER,MAAM;;EAEN;;;;;;;;EAQA;IAAkB;IAAsB;;;;;;;;EAOxC;;;;;;;;;;iBAac,iBAAiB;EAC/B;EACA;EACA;EACA;;EAEA;;;;;;iBAyBc,YACd,MAAM,KAAK;EACT;EACA;IAED;;iBAiBa,oBACd,MAAM,KAAK;EACT;EACA;IAED;UAOc;EACf;EACA;;EAEA;EACA;EACA;;EAEA,OAAO;;EAEP;IAAU;IAAe;;;UAGV;EACf;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa;;EAEb;;;;;EAKA,wBAAwB;;;;;;;;;;;;;;;KAkBd;EAEN;EACA;EACA;EACA;;EAEA,aAAa;;EAGb;EACA,SAAS;;EAGT;EACA;EACA;;EAGA;;EAEA,SAAS;EACT,UAAU,cAAc;;EAGxB;EACA,QAAQ"}
|
package/dist/wire/index.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { b as CustomTokenPricing, c as CostLedgerHandle } from "../cost-ledger-DbQdN3nO.js";
|
|
2
|
-
import {
|
|
2
|
+
import { x as ChatClient } from "../types-CBbLtr2J.js";
|
|
3
3
|
import { s as TraceStore } from "../store-BErPvYBr.js";
|
|
4
|
-
import { m as FeedbackTrajectoryStore } from "../feedback-trajectory-
|
|
4
|
+
import { m as FeedbackTrajectoryStore } from "../feedback-trajectory-CXmtITBo.js";
|
|
5
5
|
import { z } from "zod";
|
|
6
6
|
import { ServerType } from "@hono/node-server";
|
|
7
7
|
import { Hono } from "hono";
|
|
@@ -134,3 +134,17 @@ No new dependencies. No new peer deps. No `@traceai/*`, no
|
|
|
134
134
|
`@langfuse/*`, no `@opentelemetry/*` in our manifest. You bring the
|
|
135
135
|
observability stack you want; agent-eval's exporter emits the same
|
|
136
136
|
OTLP wire format independently, keyed on the endpoint you point it at.
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
## Supervisor-run resource receipts
|
|
140
|
+
|
|
141
|
+
The `/supervisor-run` reader preserves named-resource measurements in `economics.resourceRecords`.
|
|
142
|
+
Each record identifies its node and source within the normalized journal or terminal result.
|
|
143
|
+
Journal row indices refer to parsed rows after reader normalization, not original file line numbers.
|
|
144
|
+
The Markdown report renders each resource name, unit, amount, and completeness flag.
|
|
145
|
+
Comparison cells retain those same records without combining them.
|
|
146
|
+
|
|
147
|
+
A false `known` flag means the amount is a recorded subtotal, not complete usage.
|
|
148
|
+
Missing maps, explicit empty maps, and invalid fields remain distinct from measured zero.
|
|
149
|
+
Parent settlements and terminal results can include child usage, so these records are not additive totals.
|
|
150
|
+
The reporter reads evidence; it does not enforce budgets or infer missing measurements.
|
|
@@ -15,9 +15,9 @@ That would split its search state from its own selection behavior and make budge
|
|
|
15
15
|
|
|
16
16
|
An `OptimizationMethod` plugs into exactly two entry points.
|
|
17
17
|
`selfImprove()` from `/contract` is the improvement entry: one call gives the method disjoint train and selection partitions, re-scores the selected surface on a held-out split, and returns a `gateDecision`.
|
|
18
|
-
Use it
|
|
18
|
+
Use it to search for a better surface and inspect the final decision.
|
|
19
19
|
`compareOptimizationMethods()` from `/campaign` is the measurement entry: it gives every method equal inputs and scores the selected surfaces on final cases no method received.
|
|
20
|
-
Use it
|
|
20
|
+
Use it to compare selected surfaces under declared resource limits.
|
|
21
21
|
Runnable versions: [`examples/self-improve-optimizer`](../examples/self-improve-optimizer/) and [`examples/compare-optimization-methods`](../examples/compare-optimization-methods/).
|
|
22
22
|
|
|
23
23
|
## Compose searches over a candidate
|
|
@@ -75,6 +75,8 @@ Any parent receipt supplied by a custom method is also verified; it cannot repla
|
|
|
75
75
|
|
|
76
76
|
`selfImprove({ method })` executes the complete method once and measures its selected surface on final cases.
|
|
77
77
|
The method may select the unchanged baseline; that result returns `gateDecision: 'hold'` and an empty diff.
|
|
78
|
+
`winner` means the optimizer's selection, which can score worse on final cases.
|
|
79
|
+
Inspect `gateDecision` and its contributions before treating the selected surface as an improvement.
|
|
78
80
|
Agent Eval does not score train and selection cases again or choose a different surface after the method finishes.
|
|
79
81
|
|
|
80
82
|
The result type has two modes:
|
|
@@ -216,8 +218,10 @@ Agent Eval adds the run ID, evaluation count, artifact directory, source identit
|
|
|
216
218
|
- seed,
|
|
217
219
|
- campaign defaults.
|
|
218
220
|
|
|
219
|
-
|
|
221
|
+
The method input contains train and selection cases, without final test cases.
|
|
220
222
|
After every method finishes, Agent Eval scores the selected surfaces on the same final cases and reports paired lift estimates.
|
|
223
|
+
The host must also exclude final cases from callback closures, shared files, and prior optimizer state.
|
|
224
|
+
The API partition does not provide process or filesystem isolation.
|
|
221
225
|
|
|
222
226
|
```ts
|
|
223
227
|
import {
|
|
@@ -227,7 +231,7 @@ import {
|
|
|
227
231
|
} from '@tangle-network/agent-eval/campaign'
|
|
228
232
|
|
|
229
233
|
const optimizer = {
|
|
230
|
-
model:
|
|
234
|
+
model: optimizerModelId,
|
|
231
235
|
// Supplied by the package that owns execution. Discovery derives these
|
|
232
236
|
// from Runtime and one exact AgentProfile.
|
|
233
237
|
call: optimizerExecution.call,
|
|
@@ -238,10 +242,7 @@ const optimizer = {
|
|
|
238
242
|
maxRequestBytes: 2_000_000,
|
|
239
243
|
maxResponseBytes: 2_000_000,
|
|
240
244
|
maxOutputTokensPerRequest: 32_768,
|
|
241
|
-
pricing:
|
|
242
|
-
inputUsdPerMillion: Number(process.env.OPTIMIZER_INPUT_USD_PER_MILLION),
|
|
243
|
-
outputUsdPerMillion: Number(process.env.OPTIMIZER_OUTPUT_USD_PER_MILLION),
|
|
244
|
-
},
|
|
245
|
+
pricing: optimizerTokenPricing,
|
|
245
246
|
},
|
|
246
247
|
}
|
|
247
248
|
|
|
@@ -253,7 +254,7 @@ const gepa = gepaOptimizationMethod<MyCase, MyArtifact>({
|
|
|
253
254
|
kind: 'engine',
|
|
254
255
|
run: {
|
|
255
256
|
engine: 'gepa',
|
|
256
|
-
maxEvaluations:
|
|
257
|
+
maxEvaluations: 80,
|
|
257
258
|
maxProposerCostUsd: 5,
|
|
258
259
|
},
|
|
259
260
|
},
|
|
@@ -293,54 +294,48 @@ const comparison = await compareOptimizationMethods({
|
|
|
293
294
|
})
|
|
294
295
|
```
|
|
295
296
|
|
|
297
|
+
These snippets assume caller-defined cases, dispatch, judges, `optimizerExecution`, model ID, and current token pricing.
|
|
298
|
+
They illustrate configuration; the linked examples provide complete scripts.
|
|
299
|
+
Both methods above declare the same evaluation ceiling.
|
|
300
|
+
Their actual evaluations, model calls, and spend can differ.
|
|
301
|
+
|
|
296
302
|
`costCeiling` is one limit shared by optimizer-model calls, train and selection evaluations, and final test scoring.
|
|
303
|
+
It applies to calls admitted through the cost ledger.
|
|
304
|
+
Leave enough capacity for every method and the final measurements.
|
|
297
305
|
`comparison.scores` contains the final-case baseline score, selected score, lift, simultaneous interval, cost status, duration, and selected surface for each method.
|
|
298
306
|
Official method scores contain optimizer and bridge package versions, source revisions and source-tree hashes, Python runtime, custom engine module hashes, compatible run ID, exact attempt ID, resume status, evaluation count, artifact directory, and available optimizer token usage.
|
|
299
307
|
`comparison.pairwise` compares the highest-ranked method with every other method.
|
|
300
|
-
Ranking follows estimated lift
|
|
308
|
+
Ranking follows estimated lift.
|
|
309
|
+
`best` can therefore name a method whose improvement is unresolved.
|
|
310
|
+
Read each score's `decision` and the pairwise `favored` value before claiming a difference.
|
|
311
|
+
Intervals account for the method-versus-baseline contrasts and every possible method pair using a Bonferroni confidence adjustment.
|
|
312
|
+
The reported pairwise list contains only the observed best versus the alternatives.
|
|
313
|
+
|
|
314
|
+
By default, final replicates are averaged within scenarios before inference.
|
|
315
|
+
A `claim` can group scenarios into declared independent units and set `minimumEffect`.
|
|
316
|
+
Read `unitScores`, `scenarioScores`, `units`, and `pairedCellN` together to retain both units and raw observation counts.
|
|
317
|
+
Optional `finalEvidence` records fresh final-case exposure across calls sharing its ledger.
|
|
318
|
+
See [evaluation integrity](./evaluation-integrity.md) for reusable claims and their boundaries.
|
|
301
319
|
|
|
302
320
|
The runnable version is in [`examples/compare-optimization-methods`](../examples/compare-optimization-methods/).
|
|
303
321
|
|
|
304
322
|
## Install Official GEPA
|
|
305
323
|
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
```sh
|
|
309
|
-
python -m pip install agent-eval-rpc
|
|
310
|
-
python -m pip install \
|
|
311
|
-
"gepa==0.1.4" \
|
|
312
|
-
"litellm>=1.83.0,<1.92" \
|
|
313
|
-
"tqdm>=4.66.1" \
|
|
314
|
-
"cloudpickle>=3.0.0" \
|
|
315
|
-
"datasets>=2.14.6" \
|
|
316
|
-
"wandb"
|
|
317
|
-
```
|
|
318
|
-
|
|
319
|
-
Do not install `gepa[full]`; its MLflow server dependency is unpatched.
|
|
320
|
-
|
|
321
|
-
Use the tested official source revision for composed recipes and the source-only engines:
|
|
322
|
-
|
|
323
|
-
```sh
|
|
324
|
-
python -m pip install \
|
|
325
|
-
"gepa @ git+https://github.com/gepa-ai/gepa.git@f919db0a622e2e9f9204779b81fe00cc1b2d808f" \
|
|
326
|
-
"litellm>=1.83.0,<1.92" \
|
|
327
|
-
"tqdm>=4.66.1" \
|
|
328
|
-
"cloudpickle>=3.0.0" \
|
|
329
|
-
"datasets>=2.14.6" \
|
|
330
|
-
"wandb"
|
|
331
|
-
```
|
|
332
|
-
|
|
333
|
-
From this repository:
|
|
324
|
+
From the repository root, install the bridge and locked standard-engine dependencies:
|
|
334
325
|
|
|
335
326
|
```sh
|
|
336
327
|
cd clients/python
|
|
337
328
|
uv sync --frozen --group gepa-release
|
|
338
|
-
|
|
329
|
+
cd ../..
|
|
330
|
+
export OPTIMIZER_PYTHON="$PWD/clients/python/.venv/bin/python"
|
|
339
331
|
```
|
|
340
332
|
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
333
|
+
Use `--group gepa-source` instead of `--group gepa-release` for composed recipes and source-only engines.
|
|
334
|
+
Those groups select different GEPA implementations and cannot coexist in one environment.
|
|
335
|
+
Pass the Python executable as `runner.command` when configuring a method directly.
|
|
336
|
+
The runnable examples read `OPTIMIZER_PYTHON` for that setting.
|
|
337
|
+
See the [Python GEPA guide](../clients/python/README.md#gepa) for other installation paths and compatibility checks.
|
|
338
|
+
Keep dependency revisions in the Python manifest and lock rather than copying them into integration code.
|
|
344
339
|
|
|
345
340
|
## Configure GEPA
|
|
346
341
|
|
|
@@ -393,7 +388,7 @@ const method = gepaOptimizationMethod({
|
|
|
393
388
|
},
|
|
394
389
|
},
|
|
395
390
|
optimizer: {
|
|
396
|
-
model:
|
|
391
|
+
model: optimizerModelId,
|
|
397
392
|
call: optimizerExecution.call,
|
|
398
393
|
callRef: optimizerExecution.callRef,
|
|
399
394
|
budget: {
|
|
@@ -402,10 +397,7 @@ const method = gepaOptimizationMethod({
|
|
|
402
397
|
maxRequestBytes: 2_000_000,
|
|
403
398
|
maxResponseBytes: 2_000_000,
|
|
404
399
|
maxOutputTokensPerRequest: 32_768,
|
|
405
|
-
pricing:
|
|
406
|
-
inputUsdPerMillion: 0.4,
|
|
407
|
-
outputUsdPerMillion: 1.6,
|
|
408
|
-
},
|
|
400
|
+
pricing: optimizerTokenPricing,
|
|
409
401
|
},
|
|
410
402
|
},
|
|
411
403
|
describeScenario: (scenario) => ({ input: scenario.input }),
|
|
@@ -413,43 +405,28 @@ const method = gepaOptimizationMethod({
|
|
|
413
405
|
})
|
|
414
406
|
```
|
|
415
407
|
|
|
416
|
-
|
|
408
|
+
`optimizerTokenPricing` must contain the current input and output USD rates per million tokens for the selected endpoint.
|
|
417
409
|
If billed USD is unknown, omit `maxCostUsd`, `pricing`, and `maxProposerCostUsd`; the recorded cost remains unknown rather than becoming a guessed zero.
|
|
418
410
|
With `optimizer`, every recipe stage must use the standard `gepa` engine or a metered agent CLI engine (below).
|
|
419
|
-
|
|
411
|
+
The optimizer proxy receives no provider key.
|
|
412
|
+
It enforces the declared request and token limits and records the owner's usage and opaque finite JSON evidence.
|
|
420
413
|
`maxProposerCostUsd` also limits each individual GEPA engine stage.
|
|
421
414
|
|
|
422
|
-
`optimizer.call`
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
On agent-runtime, use `profileOptimizerModelCall`, which executes one exact `AgentProfile` and reports profile-digest evidence:
|
|
426
|
-
|
|
427
|
-
```ts
|
|
428
|
-
import { profileOptimizerModelCall } from '@tangle-network/agent-runtime/kernel'
|
|
429
|
-
|
|
430
|
-
const call = profileOptimizerModelCall({
|
|
431
|
-
profile: optimizerProfile,
|
|
432
|
-
context: 'prompt optimizer',
|
|
433
|
-
executor: {
|
|
434
|
-
backend: 'router',
|
|
435
|
-
routerBaseUrl: process.env.LLM_BASE_URL!,
|
|
436
|
-
routerKey: process.env.LLM_API_KEY!,
|
|
437
|
-
},
|
|
438
|
-
pricing: { inputUsdPerMillion: 0.4, outputUsdPerMillion: 1.6 },
|
|
439
|
-
})
|
|
440
|
-
```
|
|
415
|
+
`optimizer.call` supplies the model transport for this bridge.
|
|
416
|
+
Its execution owner holds the provider credentials.
|
|
441
417
|
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
418
|
+
For agent-runtime, use its maintained `profileOptimizerModelCall` adapter for the selected `AgentProfile`.
|
|
419
|
+
Keep runtime configuration in the execution-owning package.
|
|
420
|
+
For a direct endpoint, adapt [the example execution owner](../examples/_shared/openai-compatible-owner.ts) to your transport.
|
|
421
|
+
It implements `ExternalOptimizerModelCall` and returns a typed success or failure with a receipt and execution evidence.
|
|
422
|
+
The callback must resolve with that outcome; rejection loses the execution record and fails the optimizer attempt.
|
|
423
|
+
The optimizer proxy enforces its declared model limits around the supplied callback.
|
|
446
424
|
|
|
447
425
|
### Metered agent CLI engines
|
|
448
426
|
|
|
449
427
|
The `autoresearch` and `meta_harness` engines drive a `claude` CLI subprocess.
|
|
450
428
|
They ship only in the tested official source revision, not in the published `gepa` package (see [Install Official GEPA](#install-official-gepa)).
|
|
451
429
|
Set `optimizer.anthropicEndpoint: true` to admit them in proxied mode.
|
|
452
|
-
This path is measured live: a real `claude` CLI session completes with every tool call translated and every call metered.
|
|
453
430
|
The loopback proxy then also serves `POST /v1/messages` (Anthropic Messages API) and the bridge child receives `ANTHROPIC_BASE_URL`, an ephemeral `ANTHROPIC_AUTH_TOKEN`, and `ANTHROPIC_MODEL` in its environment.
|
|
454
431
|
Every CLI call becomes one canonical execution-owner call with the same reservation, receipt, and budget pipeline as reflection traffic; the run fails if the receipt count differs from the admitted call count.
|
|
455
432
|
Each agent engine run must set `engineConfig.model` to `optimizer.model`, because the engines pass `--model` and that flag beats the injected environment.
|
|
@@ -486,14 +463,14 @@ recipe: {
|
|
|
486
463
|
|
|
487
464
|
Its external model spend remains incomplete unless that engine reports it.
|
|
488
465
|
Supply provider API keys only through `runner.env`.
|
|
489
|
-
|
|
466
|
+
The child does not inherit exported provider credentials automatically.
|
|
490
467
|
The spawn builds the child environment from a fixed allowlist of benign variables (PATH, HOME, locale, `PYTHONPATH`) plus `runner.env`, so the parent environment is stripped by construction.
|
|
491
468
|
When `optimizer` is set, `removeCredentialEnvironment` also deletes credential-shaped keys from `runner.env`; the child then receives only the loopback proxy URL and an ephemeral key inside the input JSON.
|
|
492
469
|
Do not place credentials in `engineConfig` because run settings are persisted.
|
|
493
470
|
|
|
494
471
|
`describeScenario()` controls the train and selection data sent to GEPA.
|
|
495
472
|
`describeArtifact()` controls the execution evidence returned after a candidate is scored.
|
|
496
|
-
|
|
473
|
+
Agent Eval calls these callbacks only for train and selection cases; caller-owned context must respect the same boundary.
|
|
497
474
|
|
|
498
475
|
A direct standard GEPA run records `provenance.gepaCandidatePopulation`.
|
|
499
476
|
Pass that summary to `readGepaCandidatePopulationArtifact()` to verify and read every accepted candidate, its parent indices, and its selection scores.
|
|
@@ -501,42 +478,42 @@ Use `readExternalOptimizerObservationArtifact()` for every distinct callback sub
|
|
|
501
478
|
`provenance.evaluationCount` is the callback-metered evaluation total.
|
|
502
479
|
`provenance.upstreamReportedEvaluations` is GEPA's self-reported total; a difference means upstream skipped, cached, or double-counted work.
|
|
503
480
|
|
|
504
|
-
## Runtime
|
|
505
|
-
|
|
506
|
-
Each knob below has a default that works for small text campaigns and fails for agentic or slow-settling runs.
|
|
507
|
-
The table names the failure so you can set the knob before the run dies mid-spend.
|
|
481
|
+
## Runtime controls
|
|
508
482
|
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
| `timeoutMs` | 30 minutes | The whole bridge process tree is killed with `GEPA bridge exceeded 1800000ms`. An agentic run (40 evaluations over 45 s each) exceeds the default mid-spend. The same value bounds each callback POST and the runtime inspect pass. | `gepaOptimizationMethod({ timeoutMs })`, `skillOptOptimizationMethod({ timeoutMs })` |
|
|
512
|
-
| `dispatchShutdownTimeoutMs` | 5 seconds | A dispatch that cancels or settles paid calls slowly fails the cell with `CostAccountingIncompleteError` after the evaluations completed. | `runCampaign({ dispatchShutdownTimeoutMs })`; for comparisons, `compareOptimizationMethods` `optimizationRunOptions` |
|
|
513
|
-
| `servedModelPolicy` | `'exact'` | A router that substitutes a same-family model fails every proxied reflection call with a 502 `model substitution` error. `'allow-within-family'` accepts the substitute, keeps family-level claims, and forfeits per-model claims. | `optimizer.servedModelPolicy` |
|
|
514
|
-
| `reflection_lm_kwargs.num_retries` | litellm default (3) | Each failed reflection request retries 3 times inside litellm, so the proxy meters 4 request attempts per logical call and `budget.maxRequests` exhausts 4x early. Set `num_retries: 0`; the proxy already accounts each attempt. | `recipe.run.engineConfig.reflection.reflection_lm_kwargs` |
|
|
515
|
-
| `reflection_lm_kwargs.max_tokens` | `budget.maxOutputTokensPerRequest` | Every reflection request ships the full budget cap as `max_tokens`. A provider family with a lower completion cap rejects every call. A reasoning model also needs headroom for hidden reasoning tokens. Set a value at or below the smallest family cap; it must not exceed `budget.maxOutputTokensPerRequest`. | `recipe.run.engineConfig.reflection.reflection_lm_kwargs` |
|
|
516
|
-
| `maxProposerCostUsd` | unset | Without it, one engine stage can spend up to `optimizer.budget.maxCostUsd` or the campaign `costCeiling` before any limit fires. Supply it only when the execution owner can enforce billed USD. | `recipe.run.maxProposerCostUsd` |
|
|
517
|
-
| `maxEvaluations` (agent engines) | required, no default | An agent engine registers one aggregate evaluation that costs the full train set of callback evaluations. A value below the train-set size rejects mid-aggregate and GEPA records the candidate as `-inf`. | `recipe.run.maxEvaluations` |
|
|
518
|
-
| `budget.maxRequests` (agent engines) | required, no default | An agent CLI session makes tens of calls per engine run. A text-campaign-sized limit exhausts mid-run, and the CLI sees a terminal 402. | `optimizer.budget.maxRequests` |
|
|
519
|
-
| `expectUsage` | `'assert'` | A deterministic evaluator that makes no LLM calls records zero usage, so `'assert'` fails the run as a stub. Set `'off'` only for an evaluator with no paid calls. | `selfImprove({ expectUsage })` |
|
|
483
|
+
Set limits for the execution path you actually use.
|
|
484
|
+
Long-running dispatches and agent CLI engines can need different limits from short text evaluations.
|
|
520
485
|
|
|
521
|
-
|
|
486
|
+
| Setting | Default | When to change it |
|
|
487
|
+
|---|---|---|
|
|
488
|
+
| Method `timeoutMs` | 30 minutes | Bound the entire bridge run, including slow evaluations and checkpointing. |
|
|
489
|
+
| Campaign `dispatchShutdownTimeoutMs` | 5 seconds | Allow pending paid calls to settle after dispatch cancellation. |
|
|
490
|
+
| `optimizer.servedModelPolicy` | `exact` | Use `allow-within-family` only when substitutions within a model family are acceptable for the claim. |
|
|
491
|
+
| `reflection_lm_kwargs.num_retries` | Upstream setting | Set explicitly when bounding retry attempts in the GEPA reflection configuration. |
|
|
492
|
+
| `reflection_lm_kwargs.max_tokens` | Optimizer output cap | Use a limit supported by the endpoint and sufficient for the model's reasoning and output. |
|
|
493
|
+
| `recipe.run.maxProposerCostUsd` | Unset | Bound a GEPA stage separately when dollar accounting is available. |
|
|
494
|
+
| `recipe.run.maxEvaluations` | Required | Allow enough candidate-case calls for each intended aggregate evaluation. |
|
|
495
|
+
| `optimizer.budget.maxRequests` | Required | Bound optimizer calls; agent CLI sessions can use many calls per stage. |
|
|
496
|
+
| `selfImprove({ expectUsage })` | `assert` | Set `off` only for deterministic evaluation with no paid calls. |
|
|
522
497
|
|
|
523
|
-
|
|
498
|
+
Put reflection settings under `recipe.run.engineConfig.reflection.reflection_lm_kwargs` for a direct engine recipe.
|
|
499
|
+
Model substitutions remain recorded; accepting one does not establish performance of the originally requested model.
|
|
500
|
+
Request limits count calls admitted to the execution owner.
|
|
501
|
+
The owner must report its internal retries and enforce their declared bounds.
|
|
524
502
|
|
|
525
|
-
|
|
526
|
-
python -m pip install agent-eval-rpc
|
|
527
|
-
python -m pip install \
|
|
528
|
-
"skillopt @ git+https://github.com/microsoft/SkillOpt.git@61735e3922efc2b90c6d6cab561e62e98452ca90"
|
|
529
|
-
```
|
|
503
|
+
## Install Official SkillOpt
|
|
530
504
|
|
|
531
|
-
From
|
|
505
|
+
From the repository root:
|
|
532
506
|
|
|
533
507
|
```sh
|
|
534
508
|
cd clients/python
|
|
535
509
|
uv sync --frozen --group skillopt-source
|
|
510
|
+
cd ../..
|
|
511
|
+
export OPTIMIZER_PYTHON="$PWD/clients/python/.venv/bin/python"
|
|
536
512
|
```
|
|
537
513
|
|
|
538
|
-
|
|
539
|
-
|
|
514
|
+
Add `--group gepa-source` to the same sync command when comparing both methods.
|
|
515
|
+
Use the source group selected by the lock; the bridge's compatibility checks cover that implementation.
|
|
516
|
+
See the [Python SkillOpt guide](../clients/python/README.md#skillopt) for package requirements and validation.
|
|
540
517
|
|
|
541
518
|
`skillOptOptimizationMethod()` runs SkillOpt's official `ReflACTTrainer`.
|
|
542
519
|
Agent Eval supplies an environment adapter that sends each candidate and case back to the TypeScript execution and judging path.
|
|
@@ -554,28 +531,10 @@ Missing token usage, an oversized request or response, a wrong model, streaming,
|
|
|
554
531
|
|
|
555
532
|
## Use Official DSPy Optimizers
|
|
556
533
|
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
import dspy
|
|
562
|
-
|
|
563
|
-
from agent_eval_rpc import DspyJudgeMetric
|
|
564
|
-
|
|
565
|
-
dspy.configure_cache(restrict_pickle=True)
|
|
566
|
-
metric = DspyJudgeMetric(rubric_name="answer-quality")
|
|
567
|
-
gepa = dspy.GEPA(
|
|
568
|
-
metric=metric.feedback,
|
|
569
|
-
reflection_lm=dspy.LM("openai/gpt-4.1-mini"),
|
|
570
|
-
max_metric_calls=100,
|
|
571
|
-
)
|
|
572
|
-
mipro = dspy.MIPROv2(metric=metric, auto="light")
|
|
573
|
-
```
|
|
574
|
-
|
|
575
|
-
This keeps program compilation, traces, demos, and optimizer state inside DSPy.
|
|
576
|
-
Agent Eval supplies the shared rubric and returns rich feedback for `dspy.GEPA`.
|
|
577
|
-
DSPy 3.2.1 requires GEPA 0.0.27.
|
|
578
|
-
Run it in a separate Python environment from the general GEPA bridge, which uses GEPA 0.1.4.
|
|
534
|
+
Keep DSPy programs and their optimizer state inside DSPy.
|
|
535
|
+
`DspyJudgeMetric` supplies Agent Eval rubric scores and feedback to official DSPy optimizers.
|
|
536
|
+
Configure the judging client and use the [Python DSPy guide](../clients/python/README.md#dspy) for installation and examples.
|
|
537
|
+
Use a separate environment when its GEPA dependency conflicts with the general optimizer bridge.
|
|
579
538
|
|
|
580
539
|
## Resume A Compatible Run
|
|
581
540
|
|
|
@@ -637,9 +596,10 @@ const proposer: SurfaceProposer = {
|
|
|
637
596
|
Return a label and rationale when they will help later analysis.
|
|
638
597
|
Candidate creation must not read final test results.
|
|
639
598
|
|
|
640
|
-
A proposer may
|
|
599
|
+
A proposer may attach `attribution`: an opaque JSON-safe record retained on `GenerationCandidate.attribution` and in loop provenance.
|
|
641
600
|
The loop never interprets it.
|
|
642
|
-
Tag it with
|
|
601
|
+
Tag it with a schema field and validate it on readback.
|
|
602
|
+
`makePolicyEditCandidateRecord` from `/analyst` records an edit forecast that can later be compared with the measured change.
|
|
643
603
|
|
|
644
604
|
`runOptimization()` rejects a candidate whose `surfaceHash` was already admitted.
|
|
645
605
|
This includes the baseline, an earlier generation, and another candidate in the same proposal.
|
|
@@ -675,9 +635,7 @@ const result = await runOptimization({
|
|
|
675
635
|
- Final test cases may only compare surfaces after every method finishes.
|
|
676
636
|
- The same dispatch and judges score every method.
|
|
677
637
|
- Missing cost remains unknown.
|
|
678
|
-
-
|
|
679
|
-
-
|
|
638
|
+
- Bound method work before it starts, including any caller-owned operations outside the ledger.
|
|
639
|
+
- Bridge children inherit a small environment allowlist; pass unproxied provider credentials only through `runner.env`.
|
|
680
640
|
- The metered proxy path replaces provider credentials with a loopback URL and an ephemeral key.
|
|
681
641
|
- Resumed state must match every input that can change the result.
|
|
682
|
-
|
|
683
|
-
These rules make method comparisons inspectable without pretending different optimizers have identical internals.
|