@tangle-network/agent-eval 0.129.0 → 0.130.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +2 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +81 -2872
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -360
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1188
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1709
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -891
- package/dist/benchmarks/index.js +2 -60
- package/dist/benchmarks-BJgDGkAD.js +754 -0
- package/dist/benchmarks-BJgDGkAD.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6381
- package/dist/campaign/index.js +3 -213
- package/dist/campaign-aKJt6emI.js +3886 -0
- package/dist/campaign-aKJt6emI.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -175
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5565
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1938
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -33
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -618
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CD_WZ_Xr.d.ts +2250 -0
- package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index-Em67JBjs.d.ts +335 -0
- package/dist/index-Em67JBjs.d.ts.map +1 -0
- package/dist/index.d.ts +3755 -15555
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11182 -11216
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -480
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1312
- package/dist/reporting.js +6 -51
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +760 -4010
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2325 -1958
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -2087
- package/dist/rollout/index.js +8 -168
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
- package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
- package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
- package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -959
- package/dist/supervisor-run/index.js +2 -65
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -252
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1173
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/campaign-proposers.md +1 -0
- package/package.json +17 -9
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2QU3YOPR.js +0 -7374
- package/dist/chunk-2QU3YOPR.js.map +0 -1
- package/dist/chunk-3OCR4R5I.js +0 -728
- package/dist/chunk-3OCR4R5I.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-56TAVBOK.js +0 -698
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7FO3TNPI.js +0 -232
- package/dist/chunk-7FO3TNPI.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BSO5JDQH.js +0 -2335
- package/dist/chunk-BSO5JDQH.js.map +0 -1
- package/dist/chunk-C6LXANRU.js +0 -1550
- package/dist/chunk-C6LXANRU.js.map +0 -1
- package/dist/chunk-DODXQREJ.js +0 -752
- package/dist/chunk-DODXQREJ.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-E7QXT7SX.js +0 -183
- package/dist/chunk-E7QXT7SX.js.map +0 -1
- package/dist/chunk-EG66UGL4.js +0 -341
- package/dist/chunk-EG66UGL4.js.map +0 -1
- package/dist/chunk-FXTVJPYD.js +0 -576
- package/dist/chunk-FXTVJPYD.js.map +0 -1
- package/dist/chunk-G7MGMCZD.js +0 -153
- package/dist/chunk-G7MGMCZD.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H23X7XKK.js +0 -181
- package/dist/chunk-H23X7XKK.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HPWUNB47.js +0 -289
- package/dist/chunk-HPWUNB47.js.map +0 -1
- package/dist/chunk-IYCLP2N2.js +0 -766
- package/dist/chunk-IYCLP2N2.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JQSF5DQT.js +0 -701
- package/dist/chunk-JQSF5DQT.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-M4YBQKIJ.js +0 -1040
- package/dist/chunk-M4YBQKIJ.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NY44NC4A.js +0 -1056
- package/dist/chunk-NY44NC4A.js.map +0 -1
- package/dist/chunk-OIUOT4QD.js +0 -44
- package/dist/chunk-OIUOT4QD.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OWN5NPMC.js +0 -152
- package/dist/chunk-OWN5NPMC.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PC5DOSM7.js +0 -579
- package/dist/chunk-PC5DOSM7.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QB6BDBP2.js +0 -4464
- package/dist/chunk-QB6BDBP2.js.map +0 -1
- package/dist/chunk-RXHCETDZ.js +0 -536
- package/dist/chunk-RXHCETDZ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-SFLLL76A.js +0 -669
- package/dist/chunk-SFLLL76A.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-T6RLYGAD.js +0 -158
- package/dist/chunk-T6RLYGAD.js.map +0 -1
- package/dist/chunk-TJVT4QFF.js +0 -911
- package/dist/chunk-TJVT4QFF.js.map +0 -1
- package/dist/chunk-TQ7LNKZ3.js +0 -136
- package/dist/chunk-TQ7LNKZ3.js.map +0 -1
- package/dist/chunk-U4L7JRPZ.js +0 -1706
- package/dist/chunk-U4L7JRPZ.js.map +0 -1
- package/dist/chunk-U4PHLT2N.js +0 -419
- package/dist/chunk-U4PHLT2N.js.map +0 -1
- package/dist/chunk-VCZ5FQYW.js +0 -928
- package/dist/chunk-VCZ5FQYW.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WVATSFCP.js +0 -1553
- package/dist/chunk-WVATSFCP.js.map +0 -1
- package/dist/chunk-X4YIBDER.js +0 -1662
- package/dist/chunk-X4YIBDER.js.map +0 -1
- package/dist/chunk-YQN4ICPP.js +0 -355
- package/dist/chunk-YQN4ICPP.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZHTZ4EYI.js +0 -1212
- package/dist/chunk-ZHTZ4EYI.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-OJJ7CZF4.js +0 -18
- package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -1,70 +1,72 @@
|
|
|
1
|
-
import {
|
|
2
|
-
|
|
1
|
+
import { w as JudgeScore } from "../types-k9tZGKUg.js";
|
|
2
|
+
import { o as MatrixResult } from "../index-DSC51roc.js";
|
|
3
|
+
import { AgentProfile } from "@tangle-network/agent-interface";
|
|
4
|
+
//#region src/multishot/types.d.ts
|
|
3
5
|
interface MultishotMessage {
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
6
|
+
role: 'user' | 'assistant' | 'tool';
|
|
7
|
+
content: string;
|
|
8
|
+
toolCallId?: string;
|
|
9
|
+
toolCalls?: Array<{
|
|
10
|
+
id: string;
|
|
11
|
+
name: string;
|
|
12
|
+
args: Record<string, unknown>;
|
|
13
|
+
}>;
|
|
12
14
|
}
|
|
13
15
|
interface MultishotArtifact {
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
16
|
+
type: string;
|
|
17
|
+
turn: number;
|
|
18
|
+
invocation: {
|
|
19
|
+
name: string;
|
|
20
|
+
args: Record<string, unknown>;
|
|
21
|
+
};
|
|
22
|
+
content: string;
|
|
21
23
|
}
|
|
22
24
|
interface MultishotResult {
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
25
|
+
transcript: MultishotMessage[];
|
|
26
|
+
artifacts: MultishotArtifact[];
|
|
27
|
+
toolCalls: number;
|
|
28
|
+
durationMs: number;
|
|
29
|
+
costUsd: number;
|
|
28
30
|
}
|
|
29
31
|
interface MultishotToolDefinition {
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
32
|
+
type: 'function';
|
|
33
|
+
function: {
|
|
34
|
+
name: string;
|
|
35
|
+
description: string;
|
|
36
|
+
parameters: Record<string, unknown>;
|
|
37
|
+
};
|
|
36
38
|
}
|
|
37
39
|
/** One chat-completion request the multishot loop issues for a single agent
|
|
38
40
|
* (or driver) inference step. Mirrors the OpenAI-compat body the loop would
|
|
39
41
|
* otherwise POST to the Tangle router. */
|
|
40
42
|
interface MultishotTransportRequest {
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
43
|
+
model: string;
|
|
44
|
+
messages: Array<Record<string, unknown>>;
|
|
45
|
+
tools?: MultishotToolDefinition[];
|
|
46
|
+
temperature?: number;
|
|
47
|
+
maxTokens?: number;
|
|
48
|
+
signal?: AbortSignal;
|
|
47
49
|
}
|
|
48
50
|
interface MultishotTransportToolCall {
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
51
|
+
id: string;
|
|
52
|
+
type: 'function';
|
|
53
|
+
function: {
|
|
54
|
+
name: string;
|
|
55
|
+
arguments: string;
|
|
56
|
+
};
|
|
55
57
|
}
|
|
56
58
|
interface MultishotTransportResponse {
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
59
|
+
message: {
|
|
60
|
+
content?: string | null;
|
|
61
|
+
tool_calls?: MultishotTransportToolCall[];
|
|
62
|
+
};
|
|
63
|
+
usage?: {
|
|
64
|
+
prompt_tokens?: number;
|
|
65
|
+
completion_tokens?: number;
|
|
66
|
+
};
|
|
67
|
+
/** Actual spend for this call. When omitted, the loop meters cost from
|
|
68
|
+
* `usage` via the per-model router estimator (estimateRouterCost). */
|
|
69
|
+
costUsd?: number;
|
|
68
70
|
}
|
|
69
71
|
/** Execution seam for one leg of the multishot loop. When provided, it
|
|
70
72
|
* replaces the internal router HTTP call for that leg — the loop still owns
|
|
@@ -74,239 +76,129 @@ interface MultishotTransportResponse {
|
|
|
74
76
|
* signature product-side. */
|
|
75
77
|
type MultishotTransport = (req: MultishotTransportRequest) => Promise<MultishotTransportResponse>;
|
|
76
78
|
type MultishotToolExecutor = (args: Record<string, unknown>, ctx: {
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
79
|
+
apiKey: string;
|
|
80
|
+
baseUrl: string;
|
|
81
|
+
signal?: AbortSignal;
|
|
80
82
|
}) => Promise<{
|
|
81
|
-
|
|
82
|
-
|
|
83
|
+
content: string;
|
|
84
|
+
costUsd: number;
|
|
83
85
|
}>;
|
|
84
86
|
interface MultishotPersona {
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
87
|
+
/** Stable identifier — used for per-cell artifact paths + matrix axis keys. */
|
|
88
|
+
id: string;
|
|
89
|
+
/** Per-domain payload (income/profile/voice/etc.) shaped by the consumer. */
|
|
90
|
+
[k: string]: unknown;
|
|
89
91
|
}
|
|
90
92
|
interface MultishotShape<TPersona extends MultishotPersona> {
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
93
|
+
/** Opening user message (turn 0) — the persona's first ask. */
|
|
94
|
+
buildOpener: (persona: TPersona) => string;
|
|
95
|
+
/** System prompt the driver LLM uses to roleplay the persona. Should set
|
|
96
|
+
* voice, goals, constraints, time-pressure, and the "never go silent" rule. */
|
|
97
|
+
buildDriverSystemPrompt: (persona: TPersona) => string;
|
|
96
98
|
}
|
|
97
99
|
declare class MultishotDriverEmptyError extends Error {
|
|
98
|
-
|
|
99
|
-
|
|
100
|
+
readonly turn: number;
|
|
101
|
+
constructor(turn: number);
|
|
100
102
|
}
|
|
101
103
|
declare class MultishotFatalToolError extends Error {
|
|
102
|
-
|
|
104
|
+
constructor(message: string);
|
|
103
105
|
}
|
|
104
|
-
|
|
106
|
+
//#endregion
|
|
107
|
+
//#region src/multishot/router.d.ts
|
|
105
108
|
interface RouterCompletionRequest {
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
109
|
+
apiKey: string;
|
|
110
|
+
baseUrl: string;
|
|
111
|
+
model: string;
|
|
112
|
+
messages: Array<Record<string, unknown>>;
|
|
113
|
+
tools?: MultishotToolDefinition[];
|
|
114
|
+
temperature?: number;
|
|
115
|
+
maxTokens?: number;
|
|
116
|
+
signal?: AbortSignal;
|
|
114
117
|
}
|
|
115
118
|
interface RouterToolCall {
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
119
|
+
id: string;
|
|
120
|
+
type: 'function';
|
|
121
|
+
function: {
|
|
122
|
+
name: string;
|
|
123
|
+
arguments: string;
|
|
124
|
+
};
|
|
122
125
|
}
|
|
123
126
|
interface RouterCompletionResponse {
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
127
|
+
message: {
|
|
128
|
+
content?: string | null;
|
|
129
|
+
tool_calls?: RouterToolCall[];
|
|
130
|
+
};
|
|
131
|
+
usage?: {
|
|
132
|
+
prompt_tokens?: number;
|
|
133
|
+
completion_tokens?: number;
|
|
134
|
+
};
|
|
132
135
|
}
|
|
133
136
|
declare function routerCompletion(req: RouterCompletionRequest): Promise<RouterCompletionResponse>;
|
|
134
137
|
declare function estimateRouterCost(model: string, usage?: {
|
|
135
|
-
|
|
136
|
-
|
|
138
|
+
prompt_tokens?: number;
|
|
139
|
+
completion_tokens?: number;
|
|
137
140
|
}): number;
|
|
138
141
|
declare function defaultRouterBaseUrl(): string;
|
|
139
142
|
declare function requireRouterApiKey(): string;
|
|
140
|
-
|
|
143
|
+
//#endregion
|
|
144
|
+
//#region src/multishot/default-tools.d.ts
|
|
141
145
|
declare const DEFAULT_RESEARCHER_MODEL = "openai/gpt-4o-mini";
|
|
142
146
|
declare const DEFAULT_CODER_MODEL = "openai/gpt-4o-mini";
|
|
143
147
|
interface DefaultResearcherConfig {
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
+
/** Replace the system prompt to bias the researcher toward a domain's
|
|
149
|
+
* citation style. Defaults to a generic "cite sources by name" prompt. */
|
|
150
|
+
systemPrompt?: string;
|
|
151
|
+
model?: string;
|
|
148
152
|
}
|
|
149
153
|
interface DefaultCoderConfig {
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
+
/** Replace the system prompt to bias the coder toward a language /
|
|
155
|
+
* framework / artifact style. */
|
|
156
|
+
systemPrompt?: string;
|
|
157
|
+
model?: string;
|
|
154
158
|
}
|
|
155
159
|
declare const DEFAULT_DELEGATE_RESEARCH_TOOL: MultishotToolDefinition;
|
|
156
160
|
declare const DEFAULT_DELEGATE_CODE_TOOL: MultishotToolDefinition;
|
|
157
161
|
declare function createResearchExecutor(config?: DefaultResearcherConfig): MultishotToolExecutor;
|
|
158
162
|
declare function createCodeExecutor(config?: DefaultCoderConfig): MultishotToolExecutor;
|
|
159
163
|
interface DefaultToolsConfig {
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
164
|
+
research?: DefaultResearcherConfig;
|
|
165
|
+
code?: DefaultCoderConfig;
|
|
166
|
+
/** When true (default), each tool result is recorded as a typed artifact:
|
|
167
|
+
* research → type='research', code → type='code'. */
|
|
168
|
+
recordArtifacts?: boolean;
|
|
165
169
|
}
|
|
166
170
|
interface DefaultToolsBundle {
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
171
|
+
tools: MultishotToolDefinition[];
|
|
172
|
+
executors: Record<string, MultishotToolExecutor>;
|
|
173
|
+
artifactTypeFor: (toolName: string) => string | undefined;
|
|
170
174
|
}
|
|
171
175
|
declare function defaultDelegationTools(config?: DefaultToolsConfig): DefaultToolsBundle;
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
* LLM client with graceful degrade.
|
|
175
|
-
*
|
|
176
|
-
* OpenAI-compatible `/v1/chat/completions` client with:
|
|
177
|
-
* - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
|
|
178
|
-
* - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
|
|
179
|
-
* - One retry at temperature 1 when a model explicitly requires it.
|
|
180
|
-
* - Graceful json_schema → json_object degrade on 400 with schema-reject body.
|
|
181
|
-
* - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
|
|
182
|
-
* - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
|
|
183
|
-
* directly, cli-bridge subscriptions, and any router that speaks the spec.
|
|
184
|
-
*
|
|
185
|
-
* Usage:
|
|
186
|
-
* const { value, result } = await callLlmJson<MyType>(
|
|
187
|
-
* { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
|
|
188
|
-
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
189
|
-
* )
|
|
190
|
-
*
|
|
191
|
-
* `createChatClient` wraps this implementation for provider-neutral package
|
|
192
|
-
* entry points. Direct callers can use `callLlm` or `callLlmJson`.
|
|
193
|
-
*/
|
|
194
|
-
|
|
195
|
-
interface LlmUsage {
|
|
196
|
-
promptTokens: number;
|
|
197
|
-
completionTokens: number;
|
|
198
|
-
totalTokens: number;
|
|
199
|
-
/** False when the provider omitted or malformed prompt/completion usage. */
|
|
200
|
-
captured?: boolean;
|
|
201
|
-
/** Reasoning-token subset of completionTokens, when reported. */
|
|
202
|
-
reasoningTokens?: number;
|
|
203
|
-
/** Proxies populate this when prompt caching is on. */
|
|
204
|
-
cachedPromptTokens?: number;
|
|
205
|
-
}
|
|
206
|
-
interface LlmCallResult {
|
|
207
|
-
/** The text content of the first choice. Empty string if none. */
|
|
208
|
-
content: string;
|
|
209
|
-
usage: LlmUsage;
|
|
210
|
-
/**
|
|
211
|
-
* Cost in USD. Uses the provider's reported cost when present, otherwise
|
|
212
|
-
* caller-supplied token pricing. `null` when neither is available.
|
|
213
|
-
*/
|
|
214
|
-
costUsd: number | null;
|
|
215
|
-
/** Model name actually used (echoed from response). */
|
|
216
|
-
model: string;
|
|
217
|
-
/** Wall-clock duration of the HTTP call (last attempt, if retried). */
|
|
218
|
-
durationMs: number;
|
|
219
|
-
/**
|
|
220
|
-
* `finish_reason` echoed from the first choice (`stop`, `length`,
|
|
221
|
-
* `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
|
|
222
|
-
* Exposed so a free-form `callLlm` caller CAN detect a truncated answer
|
|
223
|
-
* (`length`) instead of treating a cut-off completion as complete. Note:
|
|
224
|
-
* `callLlm` does not itself reject on it — acting on this signal is the
|
|
225
|
-
* caller's responsibility (in-repo free-form drivers do not yet enforce it).
|
|
226
|
-
*/
|
|
227
|
-
finishReason?: string | null;
|
|
228
|
-
/**
|
|
229
|
-
* True when `content.trim()` is empty. An empty completion is a silent zero
|
|
230
|
-
* for free-form `callLlm` callers; this flag is the signal a caller can
|
|
231
|
-
* inspect to fail loud rather than proceed on an empty string. `callLlm`
|
|
232
|
-
* surfaces it but does not throw on it.
|
|
233
|
-
*/
|
|
234
|
-
contentEmpty?: boolean;
|
|
235
|
-
/** Raw response body. */
|
|
236
|
-
raw: Record<string, unknown>;
|
|
237
|
-
}
|
|
238
|
-
type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
|
|
239
|
-
|
|
240
|
-
/**
|
|
241
|
-
* Pass A substrate types — `runCampaign` is the one primitive every
|
|
242
|
-
* eval flow composes from. Three contracts in this file:
|
|
243
|
-
*
|
|
244
|
-
* - `Scenario` input set
|
|
245
|
-
* - `DispatchFn` how to run one scenario → artifact
|
|
246
|
-
* - `CampaignResult` defined output schema (the contract downstream tools depend on)
|
|
247
|
-
*
|
|
248
|
-
* Three more lifted from earlier substrate work (re-exported):
|
|
249
|
-
*
|
|
250
|
-
* - `JudgeConfig` pluggable dimensional scorer (0.38)
|
|
251
|
-
* - `Mutator` optimization-loop surface mutator
|
|
252
|
-
* - `Gate` promotion gate (`HeldOutGate` and friends adapt to this)
|
|
253
|
-
*
|
|
254
|
-
* No new architecture vs 0.38 — Pass A formalizes the shapes so consumers
|
|
255
|
-
* can build dashboards / CI gates / regression diffs against a stable schema.
|
|
256
|
-
*/
|
|
257
|
-
|
|
258
|
-
/** The canonical judge verdict shape — one declaration, shared by campaign
|
|
259
|
-
* judges and the multishot judge runner (which re-exports this type).
|
|
260
|
-
*
|
|
261
|
-
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
|
|
262
|
-
* multishot runner emits 0-10. Cross-scale comparison must go through
|
|
263
|
-
* `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
|
|
264
|
-
* promotion-policy) — never renormalize a producer's values in place, as
|
|
265
|
-
* downstream thresholds (`composite >= 5` in multishot/matrix.ts, live-soak
|
|
266
|
-
* `>= 7` gates) key on the producer's native scale. */
|
|
267
|
-
interface JudgeScore {
|
|
268
|
-
dimensions: Record<string, number>;
|
|
269
|
-
composite: number;
|
|
270
|
-
notes: string;
|
|
271
|
-
/** Provider metadata for display and diagnostics; accounting uses CostLedger receipts. */
|
|
272
|
-
llmCall?: LlmCallMetadata;
|
|
273
|
-
/** Set when the judge itself failed (call error, unparseable output).
|
|
274
|
-
* `composite`/`dimensions` carry no signal — aggregators MUST exclude
|
|
275
|
-
* failed scores from means instead of folding them into zeros. */
|
|
276
|
-
failed?: true;
|
|
277
|
-
/** Ensemble extras (populated by `ensembleJudge`): max per-dimension
|
|
278
|
-
* spread across surviving judges — the inter-rater signal. */
|
|
279
|
-
maxDisagreement?: number;
|
|
280
|
-
/** Ensemble extras: judge identities whose verdict failed. */
|
|
281
|
-
failedJudges?: string[];
|
|
282
|
-
/** Ensemble extras: each surviving judge's per-dimension scores. */
|
|
283
|
-
perJudge?: Record<string, Record<string, number>>;
|
|
284
|
-
}
|
|
285
|
-
|
|
176
|
+
//#endregion
|
|
177
|
+
//#region src/multishot/judges.d.ts
|
|
286
178
|
declare const DEFAULT_JUDGE_MODEL = "openai/gpt-4o-mini";
|
|
287
179
|
interface JudgeDimension {
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
180
|
+
/** JSON field name + score key. */
|
|
181
|
+
key: string;
|
|
182
|
+
/** Description shown in the judge's user prompt. */
|
|
183
|
+
description: string;
|
|
292
184
|
}
|
|
293
185
|
interface JudgeConfig<TInput> {
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
186
|
+
/** Display name (for trace + log). */
|
|
187
|
+
name: string;
|
|
188
|
+
/** Model used for this judge. */
|
|
189
|
+
model?: string;
|
|
190
|
+
/** 0-10 scored dimensions. */
|
|
191
|
+
dimensions: JudgeDimension[];
|
|
192
|
+
/** Judge system prompt — sets persona + JSON-only constraint. */
|
|
193
|
+
systemPrompt: string;
|
|
194
|
+
/** Build the user prompt from the typed input. Must include "Respond with
|
|
195
|
+
* ONLY this JSON: { ... }" listing each dimension key. */
|
|
196
|
+
buildPrompt: (input: TInput) => string;
|
|
197
|
+
/** Optional model + api overrides. */
|
|
198
|
+
apiKey?: string;
|
|
199
|
+
baseUrl?: string;
|
|
200
|
+
/** Maximum output tokens for the judge response. Defaults to 1500. */
|
|
201
|
+
maxTokens?: number;
|
|
310
202
|
}
|
|
311
203
|
declare function runJudge<TInput>(judge: JudgeConfig<TInput>, input: TInput): Promise<JudgeScore>;
|
|
312
204
|
/** Convenience: stringified dimension list for inclusion in a judge prompt.
|
|
@@ -314,212 +206,111 @@ declare function runJudge<TInput>(judge: JudgeConfig<TInput>, input: TInput): Pr
|
|
|
314
206
|
declare function renderDimensions(dims: readonly JudgeDimension[]): string;
|
|
315
207
|
/** Convenience: build the "Respond with ONLY this JSON" footer for a judge prompt. */
|
|
316
208
|
declare function renderJsonFooter(dims: readonly JudgeDimension[]): string;
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
* Validator-output verdict — substrate primitive for "did this output pass,
|
|
320
|
-
* and how well?"
|
|
321
|
-
*
|
|
322
|
-
* Used by:
|
|
323
|
-
* - `@tangle-network/agent-eval/matrix` — verdict per cell in the cartesian.
|
|
324
|
-
* - `@tangle-network/agent-runtime` — Validator<Output, Verdict = DefaultVerdict>.
|
|
325
|
-
* Runtime keeps `Validator` because it's coupled to runtime-shaped
|
|
326
|
-
* `ValidationCtx` (iteration, signal, traceEmitter); the verdict TYPE
|
|
327
|
-
* itself is a substrate concept and lives here.
|
|
328
|
-
*
|
|
329
|
-
* Repo layering: agent-eval is the substrate (no upward deps). Both
|
|
330
|
-
* agent-runtime and agent-knowledge consume this type FROM agent-eval —
|
|
331
|
-
* never the other way around. See CLAUDE.md "Repo layering" for the rule.
|
|
332
|
-
*/
|
|
333
|
-
/**
|
|
334
|
-
* Minimal verdict shape — `valid` + `score` are required; `scores` +
|
|
335
|
-
* `notes` are optional surface. Validators that need richer shapes
|
|
336
|
-
* parameterise `Validator<Output, MyVerdict>` with their own type.
|
|
337
|
-
*
|
|
338
|
-
* Need structured extras? Extend DefaultVerdict with typed fields — never
|
|
339
|
-
* serialize extras into `notes`.
|
|
340
|
-
*/
|
|
341
|
-
interface DefaultVerdict {
|
|
342
|
-
/** Whether the output meets the validator's pass criteria. */
|
|
343
|
-
valid: boolean;
|
|
344
|
-
/** Aggregate score in [0, 1]. Drivers use this for winner selection. */
|
|
345
|
-
score: number;
|
|
346
|
-
/** Per-dimension scores. Free-form; weighted into `score` by the validator. */
|
|
347
|
-
scores?: Record<string, number>;
|
|
348
|
-
/** Human-readable rationale; surfaces in trace + final-result `winner.verdict`. */
|
|
349
|
-
notes?: string;
|
|
350
|
-
}
|
|
351
|
-
|
|
352
|
-
/**
|
|
353
|
-
* N-axis cartesian matrix over substrate types — types module.
|
|
354
|
-
*
|
|
355
|
-
* The matrix is a runner + aggregator. It iterates the cartesian product of
|
|
356
|
-
* caller-provided axes (any value type — `AgentProfile` from agent-interface,
|
|
357
|
-
* `Driver` / `Validator` from agent-runtime, rubric records, thinking levels, anything)
|
|
358
|
-
* and aggregates per-axis pass/score/cost summaries. Substrate types are
|
|
359
|
-
* imported at the boundary by JSDoc only; the matrix never wraps them.
|
|
360
|
-
*/
|
|
361
|
-
|
|
362
|
-
/** A cell carries one picked value from each axis, keyed by axis name. */
|
|
363
|
-
interface MatrixCell {
|
|
364
|
-
axes: Record<string, {
|
|
365
|
-
id: string;
|
|
366
|
-
value: unknown;
|
|
367
|
-
}>;
|
|
368
|
-
/** 0-based replicate index within the same axis combination. */
|
|
369
|
-
rep: number;
|
|
370
|
-
/** Stable sort key — preserves cartesian order across concurrent execution. */
|
|
371
|
-
ordinal: number;
|
|
372
|
-
}
|
|
373
|
-
interface CellResult<Output> {
|
|
374
|
-
output: Output;
|
|
375
|
-
verdict: DefaultVerdict;
|
|
376
|
-
costUsd: number;
|
|
377
|
-
durationMs: number;
|
|
378
|
-
runId?: string;
|
|
379
|
-
/** Populated when `runCell` threw. The cell contributes 0 to passRate AND
|
|
380
|
-
* meanScore regardless of `verdict`. */
|
|
381
|
-
error?: {
|
|
382
|
-
message: string;
|
|
383
|
-
kind: string;
|
|
384
|
-
};
|
|
385
|
-
}
|
|
386
|
-
interface AxisSummary {
|
|
387
|
-
axisName: string;
|
|
388
|
-
axisValue: string;
|
|
389
|
-
cells: number;
|
|
390
|
-
passRate: number;
|
|
391
|
-
meanScore: number;
|
|
392
|
-
p50Score: number;
|
|
393
|
-
p90Score: number;
|
|
394
|
-
totalCostUsd: number;
|
|
395
|
-
meanDurationMs: number;
|
|
396
|
-
}
|
|
397
|
-
interface MatrixResult<Output> {
|
|
398
|
-
cells: Array<{
|
|
399
|
-
cell: MatrixCell;
|
|
400
|
-
runs: CellResult<Output>[];
|
|
401
|
-
}>;
|
|
402
|
-
/** `byAxis[axisName][axisValueId] = summary`. Populated only for axes
|
|
403
|
-
* named in `aggregateBy` (default = every axis in `axes`). */
|
|
404
|
-
byAxis: Record<string, Record<string, AxisSummary>>;
|
|
405
|
-
summary: {
|
|
406
|
-
totalCells: number;
|
|
407
|
-
runsExecuted: number;
|
|
408
|
-
/** Cells removed by `filter` plus cells unscheduled after the cost
|
|
409
|
-
* ceiling or abort signal tripped. */
|
|
410
|
-
cellsSkipped: number;
|
|
411
|
-
overallPassRate: number;
|
|
412
|
-
overallMeanScore: number;
|
|
413
|
-
totalCostUsd: number;
|
|
414
|
-
durationMs: number;
|
|
415
|
-
};
|
|
416
|
-
/** Stable id-like string generated at the end of the run. */
|
|
417
|
-
matrixId: string;
|
|
418
|
-
}
|
|
419
|
-
|
|
209
|
+
//#endregion
|
|
210
|
+
//#region src/multishot/matrix.d.ts
|
|
420
211
|
interface ConversationJudgeInput<TPersona extends MultishotPersona> {
|
|
421
|
-
|
|
422
|
-
|
|
212
|
+
transcript: MultishotMessage[];
|
|
213
|
+
persona: TPersona;
|
|
423
214
|
}
|
|
424
215
|
interface ArtifactJudgeInput<TPersona extends MultishotPersona> {
|
|
425
|
-
|
|
426
|
-
|
|
216
|
+
artifact: MultishotArtifact;
|
|
217
|
+
persona: TPersona;
|
|
427
218
|
}
|
|
428
219
|
interface MultishotJudges<TPersona extends MultishotPersona> {
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
220
|
+
/** Scores the full transcript end-to-end (always runs). */
|
|
221
|
+
conversation: JudgeConfig<ConversationJudgeInput<TPersona>>;
|
|
222
|
+
/** Scores each code-type artifact. Optional — omit when domain has no code artifacts. */
|
|
223
|
+
codeReview?: JudgeConfig<ArtifactJudgeInput<TPersona>>;
|
|
224
|
+
/** Scores each non-code (research/content/template) artifact. Optional. */
|
|
225
|
+
contentQuality?: JudgeConfig<ArtifactJudgeInput<TPersona>>;
|
|
226
|
+
/** Which artifact types route to codeReview. Defaults to ['code']. */
|
|
227
|
+
codeArtifactTypes?: string[];
|
|
228
|
+
/** Which artifact types route to contentQuality. Defaults to ['research']. */
|
|
229
|
+
contentArtifactTypes?: string[];
|
|
439
230
|
}
|
|
440
231
|
interface CellCompositeScore {
|
|
232
|
+
composite: number;
|
|
233
|
+
conversation: JudgeScore;
|
|
234
|
+
codeReview?: {
|
|
235
|
+
perArtifact: Array<JudgeScore & {
|
|
236
|
+
turn: number;
|
|
237
|
+
type: string;
|
|
238
|
+
}>;
|
|
441
239
|
composite: number;
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
contentQuality?: {
|
|
451
|
-
perArtifact: Array<JudgeScore & {
|
|
452
|
-
turn: number;
|
|
453
|
-
type: string;
|
|
454
|
-
}>;
|
|
455
|
-
composite: number;
|
|
456
|
-
};
|
|
240
|
+
};
|
|
241
|
+
contentQuality?: {
|
|
242
|
+
perArtifact: Array<JudgeScore & {
|
|
243
|
+
turn: number;
|
|
244
|
+
type: string;
|
|
245
|
+
}>;
|
|
246
|
+
composite: number;
|
|
247
|
+
};
|
|
457
248
|
}
|
|
458
249
|
interface RunMultishotMatrixOptions<TPersona extends MultishotPersona> {
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
250
|
+
/** AgentProfile axis (matrix primary). */
|
|
251
|
+
profiles: Array<{
|
|
252
|
+
id: string;
|
|
253
|
+
value: AgentProfile;
|
|
254
|
+
}>;
|
|
255
|
+
/** Persona axis. */
|
|
256
|
+
personas: TPersona[];
|
|
257
|
+
/** Persona-shaping callbacks. */
|
|
258
|
+
shape: MultishotShape<TPersona>;
|
|
259
|
+
/** Judge configurations. */
|
|
260
|
+
judges: MultishotJudges<TPersona>;
|
|
261
|
+
/** Tool definitions advertised to the agent. Defaults to delegate_research + delegate_code. */
|
|
262
|
+
tools?: MultishotToolDefinition[];
|
|
263
|
+
/** Map from tool name → inline executor. Must align with `tools`. */
|
|
264
|
+
toolExecutors?: Record<string, MultishotToolExecutor>;
|
|
265
|
+
/** Tool name → artifact type label. Defaults to research/code mapping. */
|
|
266
|
+
artifactTypeFor?: (toolName: string) => string | undefined;
|
|
267
|
+
/** Where per-cell artifacts land. Cells write to `<runDir>/<profileId>/<personaId>/rep-N/`. */
|
|
268
|
+
runDir: string;
|
|
269
|
+
/** Replicates per (profile, persona) cell. */
|
|
270
|
+
reps?: number;
|
|
271
|
+
/** Max conversation turns per cell. */
|
|
272
|
+
maxTurns?: number;
|
|
273
|
+
/** Maximum tool calls the agent may dispatch inside one assistant turn. */
|
|
274
|
+
maxToolDispatches?: number;
|
|
275
|
+
/** Max concurrent cells. */
|
|
276
|
+
maxConcurrency?: number;
|
|
277
|
+
/** Total $ ceiling across the matrix; cells aborted past this. */
|
|
278
|
+
costCeiling?: number;
|
|
279
|
+
/** Agent model. */
|
|
280
|
+
agentModel?: string;
|
|
281
|
+
/** Driver model. */
|
|
282
|
+
driverModel?: string;
|
|
283
|
+
/** Fallback driver models tried when the primary simulated-user model returns empty twice. */
|
|
284
|
+
driverFallbackModels?: string[];
|
|
285
|
+
/** Maximum output tokens for the first agent call in each assistant turn. */
|
|
286
|
+
agentMaxTokens?: number;
|
|
287
|
+
/** Maximum output tokens for agent follow-up calls after tool results. */
|
|
288
|
+
toolFollowupMaxTokens?: number;
|
|
289
|
+
/** Maximum output tokens for each simulated-user driver response. */
|
|
290
|
+
driverMaxTokens?: number;
|
|
291
|
+
/** Maximum output tokens for each judge response. */
|
|
292
|
+
judgeMaxTokens?: number;
|
|
293
|
+
/** Execution seam for the agent leg of every cell — replaces the router
|
|
294
|
+
* HTTP call when provided (see RunMultishotOptions.agentTransport).
|
|
295
|
+
* Judges are unaffected; configure those via MultishotJudges. */
|
|
296
|
+
agentTransport?: MultishotTransport;
|
|
297
|
+
/** Execution seam for the simulated-user driver leg of every cell. */
|
|
298
|
+
driverTransport?: MultishotTransport;
|
|
299
|
+
/** Pass-thru fields. */
|
|
300
|
+
apiKey?: string;
|
|
301
|
+
baseUrl?: string;
|
|
511
302
|
}
|
|
512
303
|
interface CellOutput {
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
304
|
+
turns: number;
|
|
305
|
+
toolCalls: number;
|
|
306
|
+
artifactCount: number;
|
|
516
307
|
}
|
|
517
308
|
interface CellCompositeInput {
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
309
|
+
conversation: JudgeScore;
|
|
310
|
+
/** Present iff the codeReview judge is configured. */
|
|
311
|
+
codeReviews?: ReadonlyArray<JudgeScore>;
|
|
312
|
+
/** Present iff the contentQuality judge is configured. */
|
|
313
|
+
contentReviews?: ReadonlyArray<JudgeScore>;
|
|
523
314
|
}
|
|
524
315
|
/** Cell composite = mean over configured judge slots, excluding failed
|
|
525
316
|
* scores: a failed conversation judge or an all-failed artifact slot carries
|
|
@@ -527,55 +318,57 @@ interface CellCompositeInput {
|
|
|
527
318
|
* configured slot failed (`allJudgesFailed` distinguishes that from a real
|
|
528
319
|
* zero). Pure — exported for deterministic testing. */
|
|
529
320
|
declare function computeCellComposite(input: CellCompositeInput): {
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
321
|
+
composite: number;
|
|
322
|
+
codeComposite: number;
|
|
323
|
+
contentComposite: number;
|
|
324
|
+
allJudgesFailed: boolean;
|
|
534
325
|
};
|
|
535
326
|
interface RunMultishotMatrixResult {
|
|
536
|
-
|
|
327
|
+
matrix: MatrixResult<CellOutput>;
|
|
537
328
|
}
|
|
538
329
|
declare function runMultishotMatrix<TPersona extends MultishotPersona>(opts: RunMultishotMatrixOptions<TPersona>): Promise<RunMultishotMatrixResult>;
|
|
539
|
-
|
|
330
|
+
//#endregion
|
|
331
|
+
//#region src/multishot/multishot.d.ts
|
|
540
332
|
interface RunMultishotOptions<TPersona extends MultishotPersona> {
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
333
|
+
profile: AgentProfile;
|
|
334
|
+
persona: TPersona;
|
|
335
|
+
shape: MultishotShape<TPersona>;
|
|
336
|
+
/** Tool definitions advertised to the agent. Defaults to delegate_research + delegate_code. */
|
|
337
|
+
tools?: MultishotToolDefinition[];
|
|
338
|
+
/** Map from tool name → executor invoked inline when the agent emits a tool_call. */
|
|
339
|
+
toolExecutors?: Record<string, MultishotToolExecutor>;
|
|
340
|
+
/** Map from tool name → artifact type label written into MultishotArtifact.type.
|
|
341
|
+
* Tools without a mapping still execute, but their results aren't surfaced as
|
|
342
|
+
* typed artifacts (only as tool messages in the transcript). */
|
|
343
|
+
artifactTypeFor?: (toolName: string) => string | undefined;
|
|
344
|
+
maxTurns?: number;
|
|
345
|
+
agentModel?: string;
|
|
346
|
+
driverModel?: string;
|
|
347
|
+
/** Fallback driver models tried when the primary simulated-user model returns empty twice. */
|
|
348
|
+
driverFallbackModels?: string[];
|
|
349
|
+
/** Maximum output tokens for the first agent call in each assistant turn. */
|
|
350
|
+
agentMaxTokens?: number;
|
|
351
|
+
/** Maximum output tokens for agent follow-up calls after tool results. */
|
|
352
|
+
toolFollowupMaxTokens?: number;
|
|
353
|
+
/** Maximum output tokens for each simulated-user driver response. */
|
|
354
|
+
driverMaxTokens?: number;
|
|
355
|
+
/** Maximum tool calls the agent may dispatch inside one assistant turn. */
|
|
356
|
+
maxToolDispatches?: number;
|
|
357
|
+
/** Execution seam for the agent leg. When provided, every agent inference
|
|
358
|
+
* step goes through this function instead of the router HTTP call; the
|
|
359
|
+
* string levers (agentModel, apiKey, baseUrl) stop applying to that leg.
|
|
360
|
+
* apiKey/baseUrl are still resolved for tool executors and any leg
|
|
361
|
+
* without an injected transport. */
|
|
362
|
+
agentTransport?: MultishotTransport;
|
|
363
|
+
/** Execution seam for the simulated-user driver leg (symmetric to
|
|
364
|
+
* agentTransport). Driver model fallback rotation still applies — the
|
|
365
|
+
* transport receives each candidate model in turn. */
|
|
366
|
+
driverTransport?: MultishotTransport;
|
|
367
|
+
apiKey?: string;
|
|
368
|
+
baseUrl?: string;
|
|
369
|
+
signal?: AbortSignal;
|
|
578
370
|
}
|
|
579
371
|
declare function runMultishot<TPersona extends MultishotPersona>(opts: RunMultishotOptions<TPersona>): Promise<MultishotResult>;
|
|
580
|
-
|
|
372
|
+
//#endregion
|
|
581
373
|
export { type ArtifactJudgeInput, type CellCompositeInput, type CellCompositeScore, type ConversationJudgeInput, DEFAULT_CODER_MODEL, DEFAULT_DELEGATE_CODE_TOOL, DEFAULT_DELEGATE_RESEARCH_TOOL, DEFAULT_JUDGE_MODEL, DEFAULT_RESEARCHER_MODEL, type DefaultCoderConfig, type DefaultResearcherConfig, type DefaultToolsBundle, type DefaultToolsConfig, type JudgeConfig, type JudgeDimension, type JudgeScore, type MultishotArtifact, MultishotDriverEmptyError, MultishotFatalToolError, type MultishotJudges, type MultishotMessage, type MultishotPersona, type MultishotResult, type MultishotShape, type MultishotToolDefinition, type MultishotToolExecutor, type MultishotTransport, type MultishotTransportRequest, type MultishotTransportResponse, type MultishotTransportToolCall, type RouterCompletionRequest, type RouterCompletionResponse, type RouterToolCall, type RunMultishotMatrixOptions, type RunMultishotMatrixResult, type RunMultishotOptions, computeCellComposite, createCodeExecutor, createResearchExecutor, defaultDelegationTools, defaultRouterBaseUrl, estimateRouterCost, renderDimensions, renderJsonFooter, requireRouterApiKey, routerCompletion, runJudge, runMultishot, runMultishotMatrix };
|
|
374
|
+
//# sourceMappingURL=index.d.ts.map
|