@tangle-network/agent-eval 0.129.0 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +1 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +81 -2872
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -360
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1188
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1709
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -891
- package/dist/benchmarks/index.js +2 -60
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6381
- package/dist/campaign/index.js +3 -213
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -175
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5565
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1938
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -33
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -618
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3755 -15555
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11182 -11216
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -480
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1312
- package/dist/reporting.js +6 -51
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +760 -4010
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2325 -1958
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -2087
- package/dist/rollout/index.js +8 -168
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -959
- package/dist/supervisor-run/index.js +2 -65
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -252
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1173
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/package.json +17 -9
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2QU3YOPR.js +0 -7374
- package/dist/chunk-2QU3YOPR.js.map +0 -1
- package/dist/chunk-3OCR4R5I.js +0 -728
- package/dist/chunk-3OCR4R5I.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-56TAVBOK.js +0 -698
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7FO3TNPI.js +0 -232
- package/dist/chunk-7FO3TNPI.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BSO5JDQH.js +0 -2335
- package/dist/chunk-BSO5JDQH.js.map +0 -1
- package/dist/chunk-C6LXANRU.js +0 -1550
- package/dist/chunk-C6LXANRU.js.map +0 -1
- package/dist/chunk-DODXQREJ.js +0 -752
- package/dist/chunk-DODXQREJ.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-E7QXT7SX.js +0 -183
- package/dist/chunk-E7QXT7SX.js.map +0 -1
- package/dist/chunk-EG66UGL4.js +0 -341
- package/dist/chunk-EG66UGL4.js.map +0 -1
- package/dist/chunk-FXTVJPYD.js +0 -576
- package/dist/chunk-FXTVJPYD.js.map +0 -1
- package/dist/chunk-G7MGMCZD.js +0 -153
- package/dist/chunk-G7MGMCZD.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H23X7XKK.js +0 -181
- package/dist/chunk-H23X7XKK.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HPWUNB47.js +0 -289
- package/dist/chunk-HPWUNB47.js.map +0 -1
- package/dist/chunk-IYCLP2N2.js +0 -766
- package/dist/chunk-IYCLP2N2.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JQSF5DQT.js +0 -701
- package/dist/chunk-JQSF5DQT.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-M4YBQKIJ.js +0 -1040
- package/dist/chunk-M4YBQKIJ.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NY44NC4A.js +0 -1056
- package/dist/chunk-NY44NC4A.js.map +0 -1
- package/dist/chunk-OIUOT4QD.js +0 -44
- package/dist/chunk-OIUOT4QD.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OWN5NPMC.js +0 -152
- package/dist/chunk-OWN5NPMC.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PC5DOSM7.js +0 -579
- package/dist/chunk-PC5DOSM7.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QB6BDBP2.js +0 -4464
- package/dist/chunk-QB6BDBP2.js.map +0 -1
- package/dist/chunk-RXHCETDZ.js +0 -536
- package/dist/chunk-RXHCETDZ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-SFLLL76A.js +0 -669
- package/dist/chunk-SFLLL76A.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-T6RLYGAD.js +0 -158
- package/dist/chunk-T6RLYGAD.js.map +0 -1
- package/dist/chunk-TJVT4QFF.js +0 -911
- package/dist/chunk-TJVT4QFF.js.map +0 -1
- package/dist/chunk-TQ7LNKZ3.js +0 -136
- package/dist/chunk-TQ7LNKZ3.js.map +0 -1
- package/dist/chunk-U4L7JRPZ.js +0 -1706
- package/dist/chunk-U4L7JRPZ.js.map +0 -1
- package/dist/chunk-U4PHLT2N.js +0 -419
- package/dist/chunk-U4PHLT2N.js.map +0 -1
- package/dist/chunk-VCZ5FQYW.js +0 -928
- package/dist/chunk-VCZ5FQYW.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WVATSFCP.js +0 -1553
- package/dist/chunk-WVATSFCP.js.map +0 -1
- package/dist/chunk-X4YIBDER.js +0 -1662
- package/dist/chunk-X4YIBDER.js.map +0 -1
- package/dist/chunk-YQN4ICPP.js +0 -355
- package/dist/chunk-YQN4ICPP.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZHTZ4EYI.js +0 -1212
- package/dist/chunk-ZHTZ4EYI.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-OJJ7CZF4.js +0 -18
- package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -1,891 +1,2 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
inputTokens: number;
|
|
4
|
-
/** Includes reasoning tokens when the provider bills them as output. */
|
|
5
|
-
outputTokens: number;
|
|
6
|
-
/** Reasoning-token subset of outputTokens, when reported. */
|
|
7
|
-
reasoningTokens?: number;
|
|
8
|
-
/** Prompt tokens served from a provider cache. */
|
|
9
|
-
cachedTokens?: number;
|
|
10
|
-
/** Prompt tokens written into a provider cache. */
|
|
11
|
-
cacheWriteTokens?: number;
|
|
12
|
-
}
|
|
13
|
-
interface CostCallBase {
|
|
14
|
-
callId: string;
|
|
15
|
-
channel: CostChannel;
|
|
16
|
-
phase: string;
|
|
17
|
-
actor: string;
|
|
18
|
-
model: string;
|
|
19
|
-
maximumCostUsd?: number;
|
|
20
|
-
tags?: Record<string, string>;
|
|
21
|
-
timestamp: number;
|
|
22
|
-
}
|
|
23
|
-
interface CostReceipt extends CostCallBase, CostUsage {
|
|
24
|
-
status: 'settled';
|
|
25
|
-
costUsd: number;
|
|
26
|
-
costUnknown: boolean;
|
|
27
|
-
usageUnknown?: boolean;
|
|
28
|
-
/** Rates used to estimate cost locally. Absent when cost is provider-reported or unknown. */
|
|
29
|
-
pricing?: {
|
|
30
|
-
inputUsdPerThousand: number;
|
|
31
|
-
cachedInputUsdPerThousand?: number;
|
|
32
|
-
cacheWriteUsdPerThousand?: number;
|
|
33
|
-
outputUsdPerThousand: number;
|
|
34
|
-
};
|
|
35
|
-
/** Cost reported by the provider, not a local token-price calculation. */
|
|
36
|
-
actualCostUsd?: number;
|
|
37
|
-
error?: string;
|
|
38
|
-
}
|
|
39
|
-
interface CostReceiptInput extends CostUsage {
|
|
40
|
-
model: string;
|
|
41
|
-
/** Caller-supplied rates for a local estimate when the provider does not report billed cost. */
|
|
42
|
-
customTokenPricing?: CustomTokenPricing;
|
|
43
|
-
actualCostUsd?: number;
|
|
44
|
-
costUnknown?: boolean;
|
|
45
|
-
usageUnknown?: boolean;
|
|
46
|
-
}
|
|
47
|
-
/** Per-million token rates for a model or endpoint not covered by package pricing. */
|
|
48
|
-
interface CustomTokenPricing {
|
|
49
|
-
/** Non-cached input tokens. */
|
|
50
|
-
inputUsdPerMillion: number;
|
|
51
|
-
/** Cache-read tokens. Falls back to the normal input rate when omitted. */
|
|
52
|
-
cachedInputUsdPerMillion?: number;
|
|
53
|
-
/** Cache-creation or cache-write tokens. Falls back to the normal input rate when omitted. */
|
|
54
|
-
cacheWriteUsdPerMillion?: number;
|
|
55
|
-
outputUsdPerMillion: number;
|
|
56
|
-
}
|
|
57
|
-
type MaximumCharge = {
|
|
58
|
-
externallyEnforcedMaximumUsd: number;
|
|
59
|
-
} | ({
|
|
60
|
-
customTokenPricing: CustomTokenPricing;
|
|
61
|
-
} & Pick<CostUsage, 'inputTokens' | 'outputTokens' | 'cachedTokens' | 'cacheWriteTokens'>) | ({
|
|
62
|
-
model: string;
|
|
63
|
-
} & CostUsage);
|
|
64
|
-
interface RunPaidCallInput<T> {
|
|
65
|
-
callId?: string;
|
|
66
|
-
channel: CostChannel;
|
|
67
|
-
phase: string;
|
|
68
|
-
actor: string;
|
|
69
|
-
/** Used before a provider receipt exists and on failures without one. */
|
|
70
|
-
model?: string;
|
|
71
|
-
tags?: Record<string, string>;
|
|
72
|
-
signal?: AbortSignal;
|
|
73
|
-
/** Provider-enforced dollar maximum, or maximum token usage with known pricing. Required when capped. */
|
|
74
|
-
maximumCharge?: MaximumCharge;
|
|
75
|
-
/** `callId` can be forwarded as the provider's idempotency key. */
|
|
76
|
-
execute(signal: AbortSignal, callId: string): Promise<T>;
|
|
77
|
-
receipt(value: T): CostReceiptInput;
|
|
78
|
-
receiptFromError?(error: Error): CostReceiptInput | undefined;
|
|
79
|
-
}
|
|
80
|
-
type PaidCallResult<T> = {
|
|
81
|
-
succeeded: true;
|
|
82
|
-
callId: string;
|
|
83
|
-
value: T;
|
|
84
|
-
receipt: CostReceipt;
|
|
85
|
-
} | {
|
|
86
|
-
succeeded: false;
|
|
87
|
-
callId?: string;
|
|
88
|
-
error: Error;
|
|
89
|
-
receipt?: CostReceipt;
|
|
90
|
-
};
|
|
91
|
-
interface ChannelRollup {
|
|
92
|
-
channel: CostChannel;
|
|
93
|
-
calls: number;
|
|
94
|
-
inputTokens: number;
|
|
95
|
-
outputTokens: number;
|
|
96
|
-
reasoningTokens?: number;
|
|
97
|
-
cachedTokens: number;
|
|
98
|
-
cacheWriteTokens?: number;
|
|
99
|
-
costUsd: number;
|
|
100
|
-
unpricedCalls: number;
|
|
101
|
-
unknownUsageCalls: number;
|
|
102
|
-
}
|
|
103
|
-
interface CostLedgerSummary {
|
|
104
|
-
totalCalls: number;
|
|
105
|
-
pendingCalls: number;
|
|
106
|
-
unresolvedCalls: number;
|
|
107
|
-
reservedCostUsd: number;
|
|
108
|
-
inputTokens: number;
|
|
109
|
-
outputTokens: number;
|
|
110
|
-
reasoningTokens?: number;
|
|
111
|
-
cachedTokens: number;
|
|
112
|
-
cacheWriteTokens?: number;
|
|
113
|
-
totalCostUsd: number;
|
|
114
|
-
byChannel: ChannelRollup[];
|
|
115
|
-
unpricedModels: string[];
|
|
116
|
-
fullyPriced: boolean;
|
|
117
|
-
usageComplete: boolean;
|
|
118
|
-
accountingComplete: boolean;
|
|
119
|
-
incompleteReasons: string[];
|
|
120
|
-
}
|
|
121
|
-
|
|
122
|
-
/**
|
|
123
|
-
* LLM client with graceful degrade.
|
|
124
|
-
*
|
|
125
|
-
* OpenAI-compatible `/v1/chat/completions` client with:
|
|
126
|
-
* - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
|
|
127
|
-
* - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
|
|
128
|
-
* - One retry at temperature 1 when a model explicitly requires it.
|
|
129
|
-
* - Graceful json_schema → json_object degrade on 400 with schema-reject body.
|
|
130
|
-
* - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
|
|
131
|
-
* - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
|
|
132
|
-
* directly, cli-bridge subscriptions, and any router that speaks the spec.
|
|
133
|
-
*
|
|
134
|
-
* Usage:
|
|
135
|
-
* const { value, result } = await callLlmJson<MyType>(
|
|
136
|
-
* { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
|
|
137
|
-
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
138
|
-
* )
|
|
139
|
-
*
|
|
140
|
-
* `createChatClient` wraps this implementation for provider-neutral package
|
|
141
|
-
* entry points. Direct callers can use `callLlm` or `callLlmJson`.
|
|
142
|
-
*/
|
|
143
|
-
|
|
144
|
-
interface LlmUsage {
|
|
145
|
-
promptTokens: number;
|
|
146
|
-
completionTokens: number;
|
|
147
|
-
totalTokens: number;
|
|
148
|
-
/** False when the provider omitted or malformed prompt/completion usage. */
|
|
149
|
-
captured?: boolean;
|
|
150
|
-
/** Reasoning-token subset of completionTokens, when reported. */
|
|
151
|
-
reasoningTokens?: number;
|
|
152
|
-
/** Proxies populate this when prompt caching is on. */
|
|
153
|
-
cachedPromptTokens?: number;
|
|
154
|
-
}
|
|
155
|
-
interface LlmCallResult {
|
|
156
|
-
/** The text content of the first choice. Empty string if none. */
|
|
157
|
-
content: string;
|
|
158
|
-
usage: LlmUsage;
|
|
159
|
-
/**
|
|
160
|
-
* Cost in USD. Uses the provider's reported cost when present, otherwise
|
|
161
|
-
* caller-supplied token pricing. `null` when neither is available.
|
|
162
|
-
*/
|
|
163
|
-
costUsd: number | null;
|
|
164
|
-
/** Model name actually used (echoed from response). */
|
|
165
|
-
model: string;
|
|
166
|
-
/** Wall-clock duration of the HTTP call (last attempt, if retried). */
|
|
167
|
-
durationMs: number;
|
|
168
|
-
/**
|
|
169
|
-
* `finish_reason` echoed from the first choice (`stop`, `length`,
|
|
170
|
-
* `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
|
|
171
|
-
* Exposed so a free-form `callLlm` caller CAN detect a truncated answer
|
|
172
|
-
* (`length`) instead of treating a cut-off completion as complete. Note:
|
|
173
|
-
* `callLlm` does not itself reject on it — acting on this signal is the
|
|
174
|
-
* caller's responsibility (in-repo free-form drivers do not yet enforce it).
|
|
175
|
-
*/
|
|
176
|
-
finishReason?: string | null;
|
|
177
|
-
/**
|
|
178
|
-
* True when `content.trim()` is empty. An empty completion is a silent zero
|
|
179
|
-
* for free-form `callLlm` callers; this flag is the signal a caller can
|
|
180
|
-
* inspect to fail loud rather than proceed on an empty string. `callLlm`
|
|
181
|
-
* surfaces it but does not throw on it.
|
|
182
|
-
*/
|
|
183
|
-
contentEmpty?: boolean;
|
|
184
|
-
/** Raw response body. */
|
|
185
|
-
raw: Record<string, unknown>;
|
|
186
|
-
}
|
|
187
|
-
type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
|
|
188
|
-
|
|
189
|
-
/**
|
|
190
|
-
* Paper-grade RunRecord schema + runtime validator.
|
|
191
|
-
*
|
|
192
|
-
* Every run that participates in a promotion gate, paper table, or
|
|
193
|
-
* researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
|
|
194
|
-
* fields are exactly those the paper "Two Loops, Three Roles" requires
|
|
195
|
-
* for reproducibility: who/what/when/cost/seed/hash, plus the search vs
|
|
196
|
-
* holdout split tag. A task score is optional because execution-only records
|
|
197
|
-
* must preserve missing labels instead of converting errors into zero quality.
|
|
198
|
-
*
|
|
199
|
-
* This is intentionally NOT a replacement for the rich `Run` /
|
|
200
|
-
* `ProposeReviewReport` / `ScenarioResult` types already in the
|
|
201
|
-
* package. Those are runtime structures with full provenance. A
|
|
202
|
-
* `RunRecord` is the analysis-time projection — the JSON-friendly
|
|
203
|
-
* row you'd put in a parquet file or paste into a notebook.
|
|
204
|
-
*
|
|
205
|
-
* Validate at the boundary:
|
|
206
|
-
*
|
|
207
|
-
* const rec = validateRunRecord(rawJson) // throws on missing
|
|
208
|
-
* const ok = isRunRecord(rawJson) // boolean check
|
|
209
|
-
* const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
|
|
210
|
-
*
|
|
211
|
-
* The validator runs in pure TS — zod is intentionally NOT a
|
|
212
|
-
* dependency. Round-trip tested in `tests/run-record.test.ts`.
|
|
213
|
-
*/
|
|
214
|
-
|
|
215
|
-
/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
|
|
216
|
-
* combined train+test pool that the optimizer is allowed to read. */
|
|
217
|
-
type RunSplitTag = 'search' | 'dev' | 'holdout';
|
|
218
|
-
interface RunTokenUsage {
|
|
219
|
-
input: number;
|
|
220
|
-
/** All generated tokens charged as output, including reasoning tokens. */
|
|
221
|
-
output: number;
|
|
222
|
-
/** Reasoning-token subset of `output`, when the provider reports it. */
|
|
223
|
-
reasoning?: number;
|
|
224
|
-
/** Prompt tokens served from a provider cache. */
|
|
225
|
-
cached?: number;
|
|
226
|
-
/** Prompt tokens written into a provider cache. */
|
|
227
|
-
cacheWrite?: number;
|
|
228
|
-
}
|
|
229
|
-
|
|
230
|
-
/**
|
|
231
|
-
* Pass A substrate types — `runCampaign` is the one primitive every
|
|
232
|
-
* eval flow composes from. Three contracts in this file:
|
|
233
|
-
*
|
|
234
|
-
* - `Scenario` input set
|
|
235
|
-
* - `DispatchFn` how to run one scenario → artifact
|
|
236
|
-
* - `CampaignResult` defined output schema (the contract downstream tools depend on)
|
|
237
|
-
*
|
|
238
|
-
* Three more lifted from earlier substrate work (re-exported):
|
|
239
|
-
*
|
|
240
|
-
* - `JudgeConfig` pluggable dimensional scorer (0.38)
|
|
241
|
-
* - `Mutator` optimization-loop surface mutator
|
|
242
|
-
* - `Gate` promotion gate (`HeldOutGate` and friends adapt to this)
|
|
243
|
-
*
|
|
244
|
-
* No new architecture vs 0.38 — Pass A formalizes the shapes so consumers
|
|
245
|
-
* can build dashboards / CI gates / regression diffs against a stable schema.
|
|
246
|
-
*/
|
|
247
|
-
|
|
248
|
-
/** Stable identifier + kind tag for any scenario. Consumers
|
|
249
|
-
* extend with their per-domain payload (persona, task, requirement, ...). */
|
|
250
|
-
interface Scenario {
|
|
251
|
-
id: string;
|
|
252
|
-
kind: string;
|
|
253
|
-
tags?: string[];
|
|
254
|
-
/**
|
|
255
|
-
* Variants with the same non-empty value receive the same seed for a given
|
|
256
|
-
* replicate. Leave unset when every scenario should have an independent seed.
|
|
257
|
-
*/
|
|
258
|
-
seedGroup?: string;
|
|
259
|
-
}
|
|
260
|
-
/** Redacted identity of a complete scenario payload retained in campaign results. */
|
|
261
|
-
interface CampaignScenarioIdentity extends Pick<Scenario, 'id' | 'kind'> {
|
|
262
|
-
scenarioDigest: `sha256:${string}`;
|
|
263
|
-
}
|
|
264
|
-
/** Context handed to every dispatch invocation. Scoped — every
|
|
265
|
-
* trace/span carries the cellId, every artifact write lands under the cell's
|
|
266
|
-
* artifact root, the cost meter accumulates per cell. */
|
|
267
|
-
interface DispatchContext {
|
|
268
|
-
cellId: string;
|
|
269
|
-
rep: number;
|
|
270
|
-
generation?: number;
|
|
271
|
-
seed: number;
|
|
272
|
-
signal: AbortSignal;
|
|
273
|
-
trace: CampaignTraceWriter;
|
|
274
|
-
artifacts: CampaignArtifactWriter;
|
|
275
|
-
cost: CampaignCostMeter;
|
|
276
|
-
/** Populated when this run is part of a multi-cycle improvement loop. */
|
|
277
|
-
cycleId?: string;
|
|
278
|
-
/** Populated when the substrate resumed from a prior cache hit. */
|
|
279
|
-
resumedFrom?: string;
|
|
280
|
-
/**
|
|
281
|
-
* Opaque placement key supplied by `RunCampaignOptions.cellPlacement`.
|
|
282
|
-
* The substrate forwards it through unchanged; placement-aware Dispatch
|
|
283
|
-
* implementations (e.g. `httpDispatch` from `/adapters/http`) read it to
|
|
284
|
-
* route the cell to the right worker / region / sandbox. `undefined`
|
|
285
|
-
* when no placement strategy is configured.
|
|
286
|
-
*/
|
|
287
|
-
placement?: string;
|
|
288
|
-
}
|
|
289
|
-
/** The canonical judge verdict shape — one declaration, shared by campaign
|
|
290
|
-
* judges and the multishot judge runner (which re-exports this type).
|
|
291
|
-
*
|
|
292
|
-
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
|
|
293
|
-
* multishot runner emits 0-10. Cross-scale comparison must go through
|
|
294
|
-
* `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
|
|
295
|
-
* promotion-policy) — never renormalize a producer's values in place, as
|
|
296
|
-
* downstream thresholds (`composite >= 5` in multishot/matrix.ts, live-soak
|
|
297
|
-
* `>= 7` gates) key on the producer's native scale. */
|
|
298
|
-
interface JudgeScore {
|
|
299
|
-
dimensions: Record<string, number>;
|
|
300
|
-
composite: number;
|
|
301
|
-
notes: string;
|
|
302
|
-
/** Provider metadata for display and diagnostics; accounting uses CostLedger receipts. */
|
|
303
|
-
llmCall?: LlmCallMetadata;
|
|
304
|
-
/** Set when the judge itself failed (call error, unparseable output).
|
|
305
|
-
* `composite`/`dimensions` carry no signal — aggregators MUST exclude
|
|
306
|
-
* failed scores from means instead of folding them into zeros. */
|
|
307
|
-
failed?: true;
|
|
308
|
-
/** Ensemble extras (populated by `ensembleJudge`): max per-dimension
|
|
309
|
-
* spread across surviving judges — the inter-rater signal. */
|
|
310
|
-
maxDisagreement?: number;
|
|
311
|
-
/** Ensemble extras: judge identities whose verdict failed. */
|
|
312
|
-
failedJudges?: string[];
|
|
313
|
-
/** Ensemble extras: each surviving judge's per-dimension scores. */
|
|
314
|
-
perJudge?: Record<string, Record<string, number>>;
|
|
315
|
-
}
|
|
316
|
-
/** Five-valued verdict taxonomy (MOSS-paper alignment). */
|
|
317
|
-
type GateDecision = 'ship' | 'hold' | 'need_more_work' | 'model_ceiling' | 'arch_ceiling';
|
|
318
|
-
/** Outcome of one check that contributed to a release decision. */
|
|
319
|
-
type GateCheckStatus = 'pass' | 'fail' | 'not_evaluated';
|
|
320
|
-
interface GateContribution {
|
|
321
|
-
name: string;
|
|
322
|
-
status: GateCheckStatus;
|
|
323
|
-
detail: unknown;
|
|
324
|
-
}
|
|
325
|
-
interface GateResult {
|
|
326
|
-
decision: GateDecision;
|
|
327
|
-
reasons: string[];
|
|
328
|
-
contributingGates: GateContribution[];
|
|
329
|
-
delta?: number;
|
|
330
|
-
}
|
|
331
|
-
/** Scoped trace writer handed to each dispatch — every span
|
|
332
|
-
* auto-tagged with the cellId so traces filter cleanly. */
|
|
333
|
-
interface CampaignTraceWriter {
|
|
334
|
-
span(name: string, attributes?: Record<string, unknown>): TraceSpan;
|
|
335
|
-
flush(): Promise<void>;
|
|
336
|
-
}
|
|
337
|
-
interface TraceSpan {
|
|
338
|
-
end(attributes?: Record<string, unknown>): void;
|
|
339
|
-
setAttribute(key: string, value: unknown): void;
|
|
340
|
-
}
|
|
341
|
-
/** Scoped artifact writer — `write(path, content)` lands under
|
|
342
|
-
* `<runDir>/<cellId>/<path>`. */
|
|
343
|
-
interface CampaignArtifactWriter {
|
|
344
|
-
write(path: string, content: string | Uint8Array): Promise<string>;
|
|
345
|
-
writeJson(path: string, value: unknown): Promise<string>;
|
|
346
|
-
}
|
|
347
|
-
/** Token usage accumulated for a cell. Aliased to the canonical `RunTokenUsage`
|
|
348
|
-
* (run-record.ts, same package) so a cell maps onto a `RunRecord` for the
|
|
349
|
-
* backend-integrity guard with ONE source of truth — a field added to
|
|
350
|
-
* `RunTokenUsage` is a compile error here, not a silent drift. */
|
|
351
|
-
type CampaignTokenUsage = RunTokenUsage;
|
|
352
|
-
/** Cell-scoped paid-call entry point. The dispatch places every paid operation
|
|
353
|
-
* inside `runPaidCall`; the returned provider result supplies one receipt with
|
|
354
|
-
* cost, tokens, and resolved model. Calls made outside this method are not
|
|
355
|
-
* admitted or captured. */
|
|
356
|
-
interface CampaignCostMeter {
|
|
357
|
-
/** The only paid-call path. Returns a typed result; callers must inspect it. */
|
|
358
|
-
runPaidCall<T>(input: Omit<RunPaidCallInput<T>, 'channel' | 'phase' | 'tags'> & {
|
|
359
|
-
channel?: CostChannel;
|
|
360
|
-
}): Promise<PaidCallResult<T>>;
|
|
361
|
-
}
|
|
362
|
-
interface CampaignCellResult<TArtifact> {
|
|
363
|
-
/** Manifest that produced this cell. Resumability refuses to reuse a cell
|
|
364
|
-
* whose manifest differs from the current run. */
|
|
365
|
-
manifestHash?: string;
|
|
366
|
-
cellId: string;
|
|
367
|
-
scenarioId: string;
|
|
368
|
-
rep: number;
|
|
369
|
-
generation?: number;
|
|
370
|
-
artifact: TArtifact;
|
|
371
|
-
judgeScores: Record<string, JudgeScore>;
|
|
372
|
-
costUsd: number;
|
|
373
|
-
/** True when at least one priced receipt used the model table instead of a provider bill. */
|
|
374
|
-
costEstimated?: boolean;
|
|
375
|
-
/** Exact durable receipts required to reuse this cached result. */
|
|
376
|
-
costCallIds?: string[];
|
|
377
|
-
/** Agent-call token usage committed by `ctx.cost.runPaidCall`.
|
|
378
|
-
* `{ input: 0, output: 0 }` when no paid agent call was recorded. */
|
|
379
|
-
tokenUsage: CampaignTokenUsage;
|
|
380
|
-
/** Concrete model from the latest committed agent receipt. Consumed by
|
|
381
|
-
* `buildRunRecord` to pin the model when the declared profile uses a
|
|
382
|
-
* runtime-resolved sentinel. */
|
|
383
|
-
resolvedModel?: string;
|
|
384
|
-
durationMs: number;
|
|
385
|
-
seed: number;
|
|
386
|
-
cached: boolean;
|
|
387
|
-
/** Stage that produced `error`. Missing on successful cells. */
|
|
388
|
-
errorStage?: 'dispatch' | 'judge';
|
|
389
|
-
/** Judge that threw when `errorStage` is `judge`. */
|
|
390
|
-
errorJudge?: string;
|
|
391
|
-
error?: string;
|
|
392
|
-
}
|
|
393
|
-
interface JudgeAggregate {
|
|
394
|
-
mean: number;
|
|
395
|
-
stdev: number;
|
|
396
|
-
ci95: [number, number];
|
|
397
|
-
n: number;
|
|
398
|
-
}
|
|
399
|
-
interface ScenarioAggregate {
|
|
400
|
-
meanComposite: number;
|
|
401
|
-
ci95: [number, number];
|
|
402
|
-
n: number;
|
|
403
|
-
}
|
|
404
|
-
interface GenerationRecord {
|
|
405
|
-
generationIndex: number;
|
|
406
|
-
candidates: GenerationCandidate[];
|
|
407
|
-
promoted: string[];
|
|
408
|
-
}
|
|
409
|
-
/** One scored candidate surface in a generation. `dimensions` + `scenarios`
|
|
410
|
-
* let a reflective proposer ground its next proposal on WHICH
|
|
411
|
-
* dimensions the candidate is weakest on and WHICH scenarios it best/worst
|
|
412
|
-
* handled — the evidence a blind `Mutator` cannot see. */
|
|
413
|
-
interface GenerationCandidate {
|
|
414
|
-
surfaceHash: string;
|
|
415
|
-
/** Mean over complete task-quality scores, or null when none were produced. */
|
|
416
|
-
composite: number | null;
|
|
417
|
-
/** Descriptive interval for `composite`, or null when no score exists. */
|
|
418
|
-
ci95: [number, number] | null;
|
|
419
|
-
/** Exact surface this candidate mutated. */
|
|
420
|
-
parentSurfaceHash?: string;
|
|
421
|
-
/** Measured search-split composite of the exact parent surface. */
|
|
422
|
-
parentComposite?: number;
|
|
423
|
-
/** Candidate composite minus its parent's composite. Present only when the
|
|
424
|
-
* candidate completed the designed denominator. */
|
|
425
|
-
observedDeltaFromParent?: number;
|
|
426
|
-
/** Whether this candidate had a scorable result for every designed campaign
|
|
427
|
-
* cell and was therefore eligible for ranking, promotion, and Pareto
|
|
428
|
-
* selection. */
|
|
429
|
-
eligibleForPromotion: boolean;
|
|
430
|
-
/** Exact denominator receipt for selection eligibility. Scores stay
|
|
431
|
-
* descriptive: an incomplete candidate is retained with its observed score
|
|
432
|
-
* and errors instead of receiving an invented penalty. */
|
|
433
|
-
coverage: {
|
|
434
|
-
expectedCells: number;
|
|
435
|
-
scorableCells: number;
|
|
436
|
-
unscorableCells: Array<{
|
|
437
|
-
cellId: string;
|
|
438
|
-
reason: string;
|
|
439
|
-
}>;
|
|
440
|
-
};
|
|
441
|
-
/** Mean score per judge dimension across all cells (scenarios × reps ×
|
|
442
|
-
* judges that reported the dimension). */
|
|
443
|
-
dimensions: Record<string, number>;
|
|
444
|
-
/** Per-scenario composite (mean over reps + judges), plus the judge's
|
|
445
|
-
* free-form `notes` for that scenario — the "why it scored low" evidence a
|
|
446
|
-
* reflective proposer grounds its next edit on. Keep `notes` GENERALIZABLE
|
|
447
|
-
* (which checks/lines/dimensions failed and how), NOT case-specific ground
|
|
448
|
-
* truth: leaking expected answers into the prompt is memorization, and the
|
|
449
|
-
* held-out gate would reject it anyway. `emitted` is a bounded excerpt of
|
|
450
|
-
* the candidate's raw output for the scenario (worst rep when reps > 1) —
|
|
451
|
-
* the "what it actually did" evidence; optional so trajectory capture is
|
|
452
|
-
* never required of a dispatch. */
|
|
453
|
-
scenarios: Array<{
|
|
454
|
-
scenarioId: string;
|
|
455
|
-
composite: number;
|
|
456
|
-
notes?: string;
|
|
457
|
-
emitted?: string;
|
|
458
|
-
}>;
|
|
459
|
-
/** Proposer-supplied short label for the change. Present when the proposer
|
|
460
|
-
* returned a `ProposedCandidate`; absent for bare-surface mutators. */
|
|
461
|
-
label?: string;
|
|
462
|
-
/** Proposer-supplied rationale — WHY this candidate was proposed. The
|
|
463
|
-
* "because rationale Z" the audit requires to survive to the result.
|
|
464
|
-
* Present when the proposer returned a `ProposedCandidate`. */
|
|
465
|
-
rationale?: string;
|
|
466
|
-
}
|
|
467
|
-
interface CampaignAggregates {
|
|
468
|
-
byJudge: Record<string, JudgeAggregate>;
|
|
469
|
-
byScenario: Record<string, ScenarioAggregate>;
|
|
470
|
-
/** Canonical campaign accounting, including worker and judge calls. */
|
|
471
|
-
cost: CostLedgerSummary;
|
|
472
|
-
/** Cells whose dispatch completed, including cells whose later judge failed. */
|
|
473
|
-
cellsExecuted: number;
|
|
474
|
-
cellsSkipped: number;
|
|
475
|
-
cellsCached: number;
|
|
476
|
-
/** All non-skipped dispatch, judge, and unclassified cell failures. */
|
|
477
|
-
cellsFailed: number;
|
|
478
|
-
/** Present on results that record failure stages. */
|
|
479
|
-
cellsDispatchFailed?: number;
|
|
480
|
-
/** Present on results that record failure stages. */
|
|
481
|
-
cellsJudgeFailed?: number;
|
|
482
|
-
/** Failures whose stage could not be classified. */
|
|
483
|
-
cellsUnclassifiedFailed?: number;
|
|
484
|
-
}
|
|
485
|
-
interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
|
|
486
|
-
/** sha256(scenarios, judges, dispatch source ref, optimizer config, seed). Stable identity for reruns. */
|
|
487
|
-
manifestHash: string;
|
|
488
|
-
/** Canonical identity of the exact scenario payloads and replicate count. */
|
|
489
|
-
splitDigest: `sha256:${string}`;
|
|
490
|
-
seed: number;
|
|
491
|
-
/** Replicates designed for every scenario in this campaign. */
|
|
492
|
-
reps: number;
|
|
493
|
-
startedAt: string;
|
|
494
|
-
endedAt: string;
|
|
495
|
-
durationMs: number;
|
|
496
|
-
cells: Array<CampaignCellResult<TArtifact>>;
|
|
497
|
-
aggregates: CampaignAggregates;
|
|
498
|
-
optimization?: {
|
|
499
|
-
generations: GenerationRecord[];
|
|
500
|
-
winnerSurfaceHash?: string;
|
|
501
|
-
};
|
|
502
|
-
gate?: GateResult;
|
|
503
|
-
prUrl?: string;
|
|
504
|
-
runDir: string;
|
|
505
|
-
artifactsByPath: Record<string, string>;
|
|
506
|
-
/** Redacted identities that let consumers verify the exact scenario payloads
|
|
507
|
-
* without retaining customer task content in the result. */
|
|
508
|
-
scenarios: Array<CampaignScenarioIdentity & Pick<TScenario, 'id' | 'kind'>>;
|
|
509
|
-
}
|
|
510
|
-
|
|
511
|
-
/**
|
|
512
|
-
* Shared types for the reference benchmark wrappers under
|
|
513
|
-
* `src/benchmarks/`. Each wrapper exports the three functions in
|
|
514
|
-
* `BenchmarkAdapter` plus its own typed `DatasetItem` shape.
|
|
515
|
-
*/
|
|
516
|
-
|
|
517
|
-
type BenchmarkTaskKind = 'retrieval' | 'rag-answer' | 'hallucination' | 'kb-improvement' | 'routing' | 'custom';
|
|
518
|
-
type BenchmarkFamily = 'beir' | 'mteb-retrieval' | 'msmarco' | 'trec-dl' | 'miracl' | 'lotte' | 'bright' | 'crag' | 'hotpotqa' | 'kilt' | 'ragtruth' | 'faithbench' | 'first-party' | 'custom';
|
|
519
|
-
interface BenchmarkDatasetItem<TPayload = unknown> {
|
|
520
|
-
/** Stable dataset-local item id (used for split assignment + paper
|
|
521
|
-
* references). Unique within a benchmark. */
|
|
522
|
-
id: string;
|
|
523
|
-
/** Free-form payload. Each benchmark defines its own shape. */
|
|
524
|
-
payload: TPayload;
|
|
525
|
-
/** Optional precomputed split. When absent, adapters use assignSplit(id). */
|
|
526
|
-
split?: RunSplitTag;
|
|
527
|
-
/** Benchmark family this row came from, e.g. `beir` or `crag`. */
|
|
528
|
-
family?: BenchmarkFamily | string;
|
|
529
|
-
/** Benchmark-local task kind, e.g. retrieval vs answer quality. */
|
|
530
|
-
taskKind?: BenchmarkTaskKind | string;
|
|
531
|
-
/** Slice labels such as language, domain, freshness, multihop, long-tail. */
|
|
532
|
-
tags?: string[];
|
|
533
|
-
/** Dataset provenance, version, URL, or license notes. */
|
|
534
|
-
source?: BenchmarkSource;
|
|
535
|
-
metadata?: Record<string, unknown>;
|
|
536
|
-
}
|
|
537
|
-
interface BenchmarkEvaluation {
|
|
538
|
-
/** [0, 1] score for the response on this item. Exact-match
|
|
539
|
-
* benchmarks use 0/1; partial-credit benchmarks may return
|
|
540
|
-
* fractional values. */
|
|
541
|
-
score: number;
|
|
542
|
-
/** Optional pass/fail projection. Defaults to `score > 0` when absent. */
|
|
543
|
-
passed?: boolean;
|
|
544
|
-
/** Numeric sub-metrics. These become report dimensions. */
|
|
545
|
-
dimensions?: Record<string, number>;
|
|
546
|
-
/** Optional bag of raw scoring signals — e.g. parsed numeric
|
|
547
|
-
* answer, regex match, judge sub-scores. */
|
|
548
|
-
raw: Record<string, unknown>;
|
|
549
|
-
notes?: string;
|
|
550
|
-
}
|
|
551
|
-
interface BenchmarkSource {
|
|
552
|
-
name?: string;
|
|
553
|
-
url?: string;
|
|
554
|
-
version?: string;
|
|
555
|
-
license?: string;
|
|
556
|
-
citation?: string;
|
|
557
|
-
}
|
|
558
|
-
/** Common signature implemented by every adapter under `src/benchmarks/*`. */
|
|
559
|
-
interface BenchmarkAdapter<_TItem = unknown, TPayload = unknown, TArtifact = string> {
|
|
560
|
-
/** Stable benchmark id such as `beir/nfcorpus` or `crag/smoke`. */
|
|
561
|
-
id?: string;
|
|
562
|
-
family?: BenchmarkFamily | string;
|
|
563
|
-
taskKind?: BenchmarkTaskKind | string;
|
|
564
|
-
description?: string;
|
|
565
|
-
source?: BenchmarkSource;
|
|
566
|
-
defaultMetric?: string;
|
|
567
|
-
/** Load the dataset for the given split. May hit the network on
|
|
568
|
-
* first call but should be cache-friendly. Adapters that don't
|
|
569
|
-
* ship the dataset itself MUST throw a clearly-marked error
|
|
570
|
-
* pointing the caller at the loader script. */
|
|
571
|
-
loadDataset(split: RunSplitTag): Promise<BenchmarkDatasetItem<TPayload>[]>;
|
|
572
|
-
/** Score a single response. Pure with respect to the inputs. */
|
|
573
|
-
evaluate(item: BenchmarkDatasetItem<TPayload>, artifact: TArtifact): Promise<BenchmarkEvaluation>;
|
|
574
|
-
/** Deterministic split assignment via item id hashing. The
|
|
575
|
-
* fraction of items in each split is implementation-defined but
|
|
576
|
-
* MUST be stable across processes and platforms. */
|
|
577
|
-
assignSplit(itemId: string): RunSplitTag;
|
|
578
|
-
}
|
|
579
|
-
interface BenchmarkScenario<TPayload = unknown> extends Scenario {
|
|
580
|
-
kind: 'benchmark';
|
|
581
|
-
benchmarkId: string;
|
|
582
|
-
family: BenchmarkFamily | string;
|
|
583
|
-
taskKind: BenchmarkTaskKind | string;
|
|
584
|
-
splitTag: RunSplitTag;
|
|
585
|
-
item: BenchmarkDatasetItem<TPayload>;
|
|
586
|
-
}
|
|
587
|
-
type BenchmarkResponder<TPayload = unknown, TArtifact = string> = (input: {
|
|
588
|
-
scenario: BenchmarkScenario<TPayload>;
|
|
589
|
-
item: BenchmarkDatasetItem<TPayload>;
|
|
590
|
-
context: DispatchContext;
|
|
591
|
-
}) => Promise<TArtifact> | TArtifact;
|
|
592
|
-
/** Split-assignment seed shared across all benchmarks. Bumping this
|
|
593
|
-
* value reshuffles every split — do NOT do that lightly. */
|
|
594
|
-
declare const BENCHMARK_SPLIT_SEED = "agent-eval-v1";
|
|
595
|
-
/**
|
|
596
|
-
* Assign an item id to one of `'search' | 'dev' | 'holdout'` using a
|
|
597
|
-
* stable 32-bit hash of `${seed}::${id}`. Default proportions:
|
|
598
|
-
*
|
|
599
|
-
* search: 60% (optimization-readable)
|
|
600
|
-
* dev: 20% (held-out for tuning, leak-on-purpose during dev)
|
|
601
|
-
* holdout:20% (paper-grade held-out, gated reads)
|
|
602
|
-
*/
|
|
603
|
-
declare function deterministicSplit(itemId: string, seed?: string): RunSplitTag;
|
|
604
|
-
|
|
605
|
-
interface BenchmarkMetricCalibrationOptions<TPayload = unknown, TArtifact = string> {
|
|
606
|
-
adapter: BenchmarkAdapter<BenchmarkDatasetItem<TPayload>, TPayload, TArtifact>;
|
|
607
|
-
item: BenchmarkDatasetItem<TPayload>;
|
|
608
|
-
weakArtifact: TArtifact;
|
|
609
|
-
strongArtifact: TArtifact;
|
|
610
|
-
maxWeakScore?: number;
|
|
611
|
-
minStrongScore?: number;
|
|
612
|
-
minGap?: number;
|
|
613
|
-
}
|
|
614
|
-
interface BenchmarkMetricCalibrationResult {
|
|
615
|
-
passed: boolean;
|
|
616
|
-
weak: BenchmarkEvaluation;
|
|
617
|
-
strong: BenchmarkEvaluation;
|
|
618
|
-
weakScore: number;
|
|
619
|
-
strongScore: number;
|
|
620
|
-
gap: number;
|
|
621
|
-
reasons: string[];
|
|
622
|
-
}
|
|
623
|
-
declare function calibrateBenchmarkMetric<TPayload = unknown, TArtifact = string>(options: BenchmarkMetricCalibrationOptions<TPayload, TArtifact>): Promise<BenchmarkMetricCalibrationResult>;
|
|
624
|
-
|
|
625
|
-
/**
|
|
626
|
-
* Synthetic routing dataset. 16 tasks across 4 categories. Used as a
|
|
627
|
-
* deterministic, dependency-free benchmark for any router that maps a
|
|
628
|
-
* natural-language request to one of a fixed set of route labels.
|
|
629
|
-
*
|
|
630
|
-
* Format (see `routing/README.md` for prose):
|
|
631
|
-
*
|
|
632
|
-
* {
|
|
633
|
-
* id: stable per-task ID (matches across processes).
|
|
634
|
-
* category: one of the four route labels.
|
|
635
|
-
* prompt: the user-facing request the router must classify.
|
|
636
|
-
* route: the ground-truth route the router should pick.
|
|
637
|
-
* synonyms: other strings that count as a correct answer.
|
|
638
|
-
* hardNegatives:close-but-wrong route labels — used to detect the
|
|
639
|
-
* "always picks the popular route" failure mode.
|
|
640
|
-
* }
|
|
641
|
-
*
|
|
642
|
-
* The four categories are intentionally cross-domain (file ops,
|
|
643
|
-
* math, search, conversation) so a router that collapses to one
|
|
644
|
-
* category is easy to spot.
|
|
645
|
-
*/
|
|
646
|
-
interface RoutingItem {
|
|
647
|
-
id: string;
|
|
648
|
-
category: 'file' | 'math' | 'search' | 'chat';
|
|
649
|
-
prompt: string;
|
|
650
|
-
/** Canonical correct route label. */
|
|
651
|
-
route: string;
|
|
652
|
-
/** Alternate route labels that also count as correct. */
|
|
653
|
-
synonyms: string[];
|
|
654
|
-
/** Wrong-but-tempting route labels (for analysis, not grading). */
|
|
655
|
-
hardNegatives: string[];
|
|
656
|
-
}
|
|
657
|
-
declare const ROUTING_DATASET: RoutingItem[];
|
|
658
|
-
|
|
659
|
-
/**
|
|
660
|
-
* Routing benchmark — synthetic, dependency-free, ships in the
|
|
661
|
-
* package. 16 cross-category items in `dataset.ts`. See
|
|
662
|
-
* `routing/README.md` for the format.
|
|
663
|
-
*
|
|
664
|
-
* `evaluate` does case-insensitive exact match against the canonical
|
|
665
|
-
* route plus declared synonyms. The first valid route token in the
|
|
666
|
-
* response wins; everything else is ignored. Wrong answers also
|
|
667
|
-
* report whether they hit a hard negative — useful when triaging
|
|
668
|
-
* "always picks the popular route" failure modes.
|
|
669
|
-
*/
|
|
670
|
-
|
|
671
|
-
type RoutingPayload = RoutingItem;
|
|
672
|
-
type RoutingDatasetItem = BenchmarkDatasetItem<RoutingPayload>;
|
|
673
|
-
declare class RoutingAdapter implements BenchmarkAdapter<RoutingDatasetItem, RoutingPayload> {
|
|
674
|
-
readonly id = "first-party/routing";
|
|
675
|
-
readonly family = "first-party";
|
|
676
|
-
readonly taskKind = "routing";
|
|
677
|
-
readonly description = "Synthetic fixed-route classification smoke benchmark";
|
|
678
|
-
readonly defaultMetric = "route_exact_match";
|
|
679
|
-
loadDataset(split: RunSplitTag): Promise<RoutingDatasetItem[]>;
|
|
680
|
-
evaluate(item: RoutingDatasetItem, response: string): Promise<BenchmarkEvaluation>;
|
|
681
|
-
assignSplit(itemId: string): RunSplitTag;
|
|
682
|
-
}
|
|
683
|
-
/**
|
|
684
|
-
* Pull route-shaped tokens out of a model response. Routes look like
|
|
685
|
-
* `category.action` (`fs.write`, `chat.reply`). Bare alphanumerics
|
|
686
|
-
* are not routes, but `category.action` patterns are robust to most
|
|
687
|
-
* model wrappers (JSON output, prose explanations, code fences).
|
|
688
|
-
*/
|
|
689
|
-
declare function extractRouteTokens(response: string): string[];
|
|
690
|
-
declare const loadDataset: (split: RunSplitTag) => Promise<RoutingDatasetItem[]>;
|
|
691
|
-
declare const evaluate: (item: RoutingDatasetItem, response: string) => Promise<BenchmarkEvaluation>;
|
|
692
|
-
declare const assignSplit: (itemId: string) => RunSplitTag;
|
|
693
|
-
|
|
694
|
-
declare const index_ROUTING_DATASET: typeof ROUTING_DATASET;
|
|
695
|
-
type index_RoutingAdapter = RoutingAdapter;
|
|
696
|
-
declare const index_RoutingAdapter: typeof RoutingAdapter;
|
|
697
|
-
type index_RoutingDatasetItem = RoutingDatasetItem;
|
|
698
|
-
type index_RoutingItem = RoutingItem;
|
|
699
|
-
type index_RoutingPayload = RoutingPayload;
|
|
700
|
-
declare const index_assignSplit: typeof assignSplit;
|
|
701
|
-
declare const index_evaluate: typeof evaluate;
|
|
702
|
-
declare const index_extractRouteTokens: typeof extractRouteTokens;
|
|
703
|
-
declare const index_loadDataset: typeof loadDataset;
|
|
704
|
-
declare namespace index {
|
|
705
|
-
export { index_ROUTING_DATASET as ROUTING_DATASET, index_RoutingAdapter as RoutingAdapter, type index_RoutingDatasetItem as RoutingDatasetItem, type index_RoutingItem as RoutingItem, type index_RoutingPayload as RoutingPayload, index_assignSplit as assignSplit, index_evaluate as evaluate, index_extractRouteTokens as extractRouteTokens, index_loadDataset as loadDataset };
|
|
706
|
-
}
|
|
707
|
-
|
|
708
|
-
/**
|
|
709
|
-
* `CampaignStorage` — the filesystem seam `runCampaign` writes through
|
|
710
|
-
* (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).
|
|
711
|
-
*
|
|
712
|
-
* The default (`fsCampaignStorage`) is the Node filesystem — identical
|
|
713
|
-
* behavior to the inline `node:fs` calls it replaces, so existing CLI
|
|
714
|
-
* consumers are unaffected. `inMemoryCampaignStorage` keeps everything in a
|
|
715
|
-
* `Map`, so the substrate runs in environments WITHOUT a filesystem
|
|
716
|
-
* (Cloudflare Workers, Deno Deploy, other edge runtimes) — the campaign
|
|
717
|
-
* still produces its `CampaignResult` (cells + aggregates) in memory;
|
|
718
|
-
* artifacts/traces simply aren't persisted to disk.
|
|
719
|
-
*
|
|
720
|
-
* Paths are opaque keys to the in-memory adapter — it does not parse them,
|
|
721
|
-
* so the same `join(...)`-built paths work unchanged across both adapters.
|
|
722
|
-
*/
|
|
723
|
-
interface CampaignStorage {
|
|
724
|
-
/** Ensure a directory exists (recursive). No-op for in-memory. */
|
|
725
|
-
ensureDir(dir: string): void;
|
|
726
|
-
/** Does this path exist (as a written file or an ensured dir)? */
|
|
727
|
-
exists(path: string): boolean;
|
|
728
|
-
/** Read a UTF-8 file; `undefined` when missing or unreadable. */
|
|
729
|
-
read(path: string): string | undefined;
|
|
730
|
-
/** Write a file (string or bytes). Parent dir is assumed ensured. */
|
|
731
|
-
write(path: string, content: string | Uint8Array): void;
|
|
732
|
-
/** Append only when the current UTF-8 byte length matches `expectedBytes`.
|
|
733
|
-
* Returns the new length, or undefined when another writer won. */
|
|
734
|
-
append(path: string, content: string, expectedBytes: number): number | undefined;
|
|
735
|
-
}
|
|
736
|
-
|
|
737
|
-
interface BenchmarkRunOptions<TPayload = unknown, TArtifact = string> {
|
|
738
|
-
adapter: BenchmarkAdapter<BenchmarkDatasetItem<TPayload>, TPayload, TArtifact>;
|
|
739
|
-
respond: BenchmarkResponder<TPayload, TArtifact>;
|
|
740
|
-
splits?: readonly RunSplitTag[];
|
|
741
|
-
runDir: string;
|
|
742
|
-
repo?: string;
|
|
743
|
-
seed?: number;
|
|
744
|
-
reps?: number;
|
|
745
|
-
resumable?: boolean;
|
|
746
|
-
costCeiling?: number;
|
|
747
|
-
maxConcurrency?: number;
|
|
748
|
-
dispatchTimeoutMs?: number;
|
|
749
|
-
expectUsage?: 'assert' | 'warn' | 'off';
|
|
750
|
-
storage?: CampaignStorage;
|
|
751
|
-
now?: () => Date;
|
|
752
|
-
}
|
|
753
|
-
interface BenchmarkReport {
|
|
754
|
-
benchmarkId: string;
|
|
755
|
-
family: BenchmarkFamily | string;
|
|
756
|
-
taskKind: BenchmarkTaskKind | string;
|
|
757
|
-
source?: BenchmarkAdapter['source'];
|
|
758
|
-
runDir: string;
|
|
759
|
-
manifestHash: string;
|
|
760
|
-
seed: number;
|
|
761
|
-
startedAt: string;
|
|
762
|
-
endedAt: string;
|
|
763
|
-
durationMs: number;
|
|
764
|
-
totalItems: number;
|
|
765
|
-
totalCells: number;
|
|
766
|
-
cellsFailed: number;
|
|
767
|
-
cellsCached: number;
|
|
768
|
-
totalCostUsd: number;
|
|
769
|
-
splits: Record<string, BenchmarkSliceSummary>;
|
|
770
|
-
tags: Record<string, BenchmarkSliceSummary>;
|
|
771
|
-
dimensions: Record<string, BenchmarkDistribution>;
|
|
772
|
-
score: BenchmarkDistribution;
|
|
773
|
-
costUsd: BenchmarkDistribution;
|
|
774
|
-
latencyMs: BenchmarkDistribution;
|
|
775
|
-
}
|
|
776
|
-
interface BenchmarkSliceSummary {
|
|
777
|
-
n: number;
|
|
778
|
-
meanScore: number;
|
|
779
|
-
passRate: number;
|
|
780
|
-
score: BenchmarkDistribution;
|
|
781
|
-
costUsd: BenchmarkDistribution;
|
|
782
|
-
latencyMs: BenchmarkDistribution;
|
|
783
|
-
}
|
|
784
|
-
interface BenchmarkDistribution {
|
|
785
|
-
n: number;
|
|
786
|
-
min: number;
|
|
787
|
-
mean: number;
|
|
788
|
-
median: number;
|
|
789
|
-
p90: number;
|
|
790
|
-
max: number;
|
|
791
|
-
}
|
|
792
|
-
interface BenchmarkRunResult<TPayload = unknown, TArtifact = string> {
|
|
793
|
-
scenarios: Array<BenchmarkScenario<TPayload>>;
|
|
794
|
-
campaign: CampaignResult<TArtifact, BenchmarkScenario<TPayload>>;
|
|
795
|
-
report: BenchmarkReport;
|
|
796
|
-
reportJsonPath: string;
|
|
797
|
-
reportMarkdownPath: string;
|
|
798
|
-
}
|
|
799
|
-
declare function runBenchmarkAdapter<TPayload = unknown, TArtifact = string>(options: BenchmarkRunOptions<TPayload, TArtifact>): Promise<BenchmarkRunResult<TPayload, TArtifact>>;
|
|
800
|
-
declare function summarizeBenchmarkCampaign<TPayload, TArtifact>(input: {
|
|
801
|
-
adapter: BenchmarkAdapter<BenchmarkDatasetItem<TPayload>, TPayload, TArtifact>;
|
|
802
|
-
scenarios: Array<BenchmarkScenario<TPayload>>;
|
|
803
|
-
campaign: CampaignResult<TArtifact, BenchmarkScenario<TPayload>>;
|
|
804
|
-
}): BenchmarkReport;
|
|
805
|
-
declare function renderBenchmarkReportMarkdown(report: BenchmarkReport): string;
|
|
806
|
-
|
|
807
|
-
interface StandardRetrievalDocument {
|
|
808
|
-
id: string;
|
|
809
|
-
title?: string;
|
|
810
|
-
text: string;
|
|
811
|
-
metadata?: Record<string, unknown>;
|
|
812
|
-
}
|
|
813
|
-
interface StandardRetrievalQuery {
|
|
814
|
-
id: string;
|
|
815
|
-
text: string;
|
|
816
|
-
metadata?: Record<string, unknown>;
|
|
817
|
-
}
|
|
818
|
-
interface StandardRetrievalQrel {
|
|
819
|
-
queryId: string;
|
|
820
|
-
documentId: string;
|
|
821
|
-
score: number;
|
|
822
|
-
}
|
|
823
|
-
interface StandardRetrievalPayload {
|
|
824
|
-
queryId: string;
|
|
825
|
-
query: string;
|
|
826
|
-
expectedDocumentIds: string[];
|
|
827
|
-
expectedScores: Record<string, number>;
|
|
828
|
-
corpus?: Record<string, StandardRetrievalDocument>;
|
|
829
|
-
metadata?: Record<string, unknown>;
|
|
830
|
-
}
|
|
831
|
-
interface BuildStandardRetrievalItemsOptions {
|
|
832
|
-
benchmarkId: string;
|
|
833
|
-
family: BenchmarkFamily | string;
|
|
834
|
-
queries: readonly StandardRetrievalQuery[];
|
|
835
|
-
qrels: readonly StandardRetrievalQrel[];
|
|
836
|
-
corpus?: readonly StandardRetrievalDocument[];
|
|
837
|
-
includeCorpusInPayload?: boolean;
|
|
838
|
-
source?: BenchmarkSource;
|
|
839
|
-
tags?: readonly string[];
|
|
840
|
-
splitOf?: (queryId: string) => RunSplitTag;
|
|
841
|
-
}
|
|
842
|
-
interface RetrievalIdAdapterOptions extends BuildStandardRetrievalItemsOptions {
|
|
843
|
-
responseIdPattern?: RegExp;
|
|
844
|
-
cutoffs?: readonly number[];
|
|
845
|
-
primaryMetric?: string;
|
|
846
|
-
passMetric?: string;
|
|
847
|
-
passThreshold?: number;
|
|
848
|
-
}
|
|
849
|
-
interface StandardRetrievalResult {
|
|
850
|
-
id?: string;
|
|
851
|
-
documentId?: string;
|
|
852
|
-
docId?: string;
|
|
853
|
-
score?: number;
|
|
854
|
-
}
|
|
855
|
-
type StandardRetrievalArtifact = string | readonly string[] | readonly StandardRetrievalResult[] | {
|
|
856
|
-
ids?: readonly string[];
|
|
857
|
-
documentIds?: readonly string[];
|
|
858
|
-
results?: readonly StandardRetrievalResult[];
|
|
859
|
-
};
|
|
860
|
-
interface StandardRetrievalEvaluationOptions {
|
|
861
|
-
responseIdPattern?: RegExp;
|
|
862
|
-
cutoffs?: readonly number[];
|
|
863
|
-
primaryMetric?: string;
|
|
864
|
-
passMetric?: string;
|
|
865
|
-
passThreshold?: number;
|
|
866
|
-
}
|
|
867
|
-
declare function parseJsonlRows<T = unknown>(text: string): T[];
|
|
868
|
-
declare function parseTsvRows(text: string): string[][];
|
|
869
|
-
declare function parseQrels(text: string): StandardRetrievalQrel[];
|
|
870
|
-
declare function parseBeirCorpusJsonl(text: string): StandardRetrievalDocument[];
|
|
871
|
-
declare function parseBeirQueriesJsonl(text: string): StandardRetrievalQuery[];
|
|
872
|
-
declare function buildStandardRetrievalItems(options: BuildStandardRetrievalItemsOptions): Array<BenchmarkDatasetItem<StandardRetrievalPayload>>;
|
|
873
|
-
declare function createRetrievalIdBenchmarkAdapter(options: RetrievalIdAdapterOptions): BenchmarkAdapter<BenchmarkDatasetItem<StandardRetrievalPayload>, StandardRetrievalPayload, StandardRetrievalArtifact>;
|
|
874
|
-
declare function evaluateStandardRetrieval(payload: StandardRetrievalPayload, artifact: StandardRetrievalArtifact, options?: StandardRetrievalEvaluationOptions): {
|
|
875
|
-
score: number;
|
|
876
|
-
passed: boolean;
|
|
877
|
-
dimensions: Record<string, number>;
|
|
878
|
-
raw: {
|
|
879
|
-
rankedDocumentIds: string[];
|
|
880
|
-
expectedDocumentIds: string[];
|
|
881
|
-
expectedScores: Record<string, number>;
|
|
882
|
-
};
|
|
883
|
-
};
|
|
884
|
-
declare function normalizeRetrievedDocumentIds(artifact: StandardRetrievalArtifact, responseIdPattern?: RegExp): string[];
|
|
885
|
-
declare function retrievalMetricsAtCutoff(input: {
|
|
886
|
-
rankedDocumentIds: readonly string[];
|
|
887
|
-
expectedScores: Record<string, number>;
|
|
888
|
-
cutoff: number;
|
|
889
|
-
}): Record<string, number>;
|
|
890
|
-
|
|
891
|
-
export { BENCHMARK_SPLIT_SEED, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkDistribution, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkMetricCalibrationOptions, type BenchmarkMetricCalibrationResult, type BenchmarkReport, type BenchmarkResponder, type BenchmarkRunOptions, type BenchmarkRunResult, type BenchmarkScenario, type BenchmarkSliceSummary, type BenchmarkSource, type BenchmarkTaskKind, type BuildStandardRetrievalItemsOptions, type RetrievalIdAdapterOptions, type StandardRetrievalArtifact, type StandardRetrievalDocument, type StandardRetrievalEvaluationOptions, type StandardRetrievalPayload, type StandardRetrievalQrel, type StandardRetrievalQuery, type StandardRetrievalResult, buildStandardRetrievalItems, calibrateBenchmarkMetric, createRetrievalIdBenchmarkAdapter, deterministicSplit, evaluateStandardRetrieval, normalizeRetrievedDocumentIds, parseBeirCorpusJsonl, parseBeirQueriesJsonl, parseJsonlRows, parseQrels, parseTsvRows, renderBenchmarkReportMarkdown, retrievalMetricsAtCutoff, index as routing, runBenchmarkAdapter, summarizeBenchmarkCampaign };
|
|
1
|
+
import { A as BenchmarkMetricCalibrationOptions, B as BenchmarkSource, C as BenchmarkRunOptions, D as runBenchmarkAdapter, E as renderBenchmarkReportMarkdown, F as BenchmarkDatasetItem, H as deterministicSplit, I as BenchmarkEvaluation, L as BenchmarkFamily, M as calibrateBenchmarkMetric, N as BENCHMARK_SPLIT_SEED, O as summarizeBenchmarkCampaign, P as BenchmarkAdapter, R as BenchmarkResponder, S as BenchmarkReport, T as BenchmarkSliceSummary, V as BenchmarkTaskKind, _ as parseJsonlRows, a as StandardRetrievalDocument, b as retrievalMetricsAtCutoff, c as StandardRetrievalQrel, d as buildStandardRetrievalItems, f as createRetrievalIdBenchmarkAdapter, g as parseBeirQueriesJsonl, h as parseBeirCorpusJsonl, i as StandardRetrievalArtifact, j as BenchmarkMetricCalibrationResult, k as index_d_exports, l as StandardRetrievalQuery, m as normalizeRetrievedDocumentIds, n as BuildStandardRetrievalItemsOptions, o as StandardRetrievalEvaluationOptions, p as evaluateStandardRetrieval, r as RetrievalIdAdapterOptions, s as StandardRetrievalPayload, u as StandardRetrievalResult, v as parseQrels, w as BenchmarkRunResult, x as BenchmarkDistribution, y as parseTsvRows, z as BenchmarkScenario } from "../index-CAPUUKaM.js";
|
|
2
|
+
export { BENCHMARK_SPLIT_SEED, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkDistribution, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkMetricCalibrationOptions, type BenchmarkMetricCalibrationResult, type BenchmarkReport, type BenchmarkResponder, type BenchmarkRunOptions, type BenchmarkRunResult, type BenchmarkScenario, type BenchmarkSliceSummary, type BenchmarkSource, type BenchmarkTaskKind, type BuildStandardRetrievalItemsOptions, type RetrievalIdAdapterOptions, type StandardRetrievalArtifact, type StandardRetrievalDocument, type StandardRetrievalEvaluationOptions, type StandardRetrievalPayload, type StandardRetrievalQrel, type StandardRetrievalQuery, type StandardRetrievalResult, buildStandardRetrievalItems, calibrateBenchmarkMetric, createRetrievalIdBenchmarkAdapter, deterministicSplit, evaluateStandardRetrieval, normalizeRetrievedDocumentIds, parseBeirCorpusJsonl, parseBeirQueriesJsonl, parseJsonlRows, parseQrels, parseTsvRows, renderBenchmarkReportMarkdown, retrievalMetricsAtCutoff, index_d_exports as routing, runBenchmarkAdapter, summarizeBenchmarkCampaign };
|