@tangle-network/agent-eval 0.129.0 → 0.130.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +2 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +81 -2872
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -360
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1188
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1709
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -891
- package/dist/benchmarks/index.js +2 -60
- package/dist/benchmarks-BJgDGkAD.js +754 -0
- package/dist/benchmarks-BJgDGkAD.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6381
- package/dist/campaign/index.js +3 -213
- package/dist/campaign-aKJt6emI.js +3886 -0
- package/dist/campaign-aKJt6emI.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -175
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5565
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1938
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -33
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -618
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CD_WZ_Xr.d.ts +2250 -0
- package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index-Em67JBjs.d.ts +335 -0
- package/dist/index-Em67JBjs.d.ts.map +1 -0
- package/dist/index.d.ts +3755 -15555
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11182 -11216
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -480
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1312
- package/dist/reporting.js +6 -51
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +760 -4010
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2325 -1958
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -2087
- package/dist/rollout/index.js +8 -168
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
- package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
- package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
- package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -959
- package/dist/supervisor-run/index.js +2 -65
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -252
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1173
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/campaign-proposers.md +1 -0
- package/package.json +17 -9
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2QU3YOPR.js +0 -7374
- package/dist/chunk-2QU3YOPR.js.map +0 -1
- package/dist/chunk-3OCR4R5I.js +0 -728
- package/dist/chunk-3OCR4R5I.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-56TAVBOK.js +0 -698
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7FO3TNPI.js +0 -232
- package/dist/chunk-7FO3TNPI.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BSO5JDQH.js +0 -2335
- package/dist/chunk-BSO5JDQH.js.map +0 -1
- package/dist/chunk-C6LXANRU.js +0 -1550
- package/dist/chunk-C6LXANRU.js.map +0 -1
- package/dist/chunk-DODXQREJ.js +0 -752
- package/dist/chunk-DODXQREJ.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-E7QXT7SX.js +0 -183
- package/dist/chunk-E7QXT7SX.js.map +0 -1
- package/dist/chunk-EG66UGL4.js +0 -341
- package/dist/chunk-EG66UGL4.js.map +0 -1
- package/dist/chunk-FXTVJPYD.js +0 -576
- package/dist/chunk-FXTVJPYD.js.map +0 -1
- package/dist/chunk-G7MGMCZD.js +0 -153
- package/dist/chunk-G7MGMCZD.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H23X7XKK.js +0 -181
- package/dist/chunk-H23X7XKK.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HPWUNB47.js +0 -289
- package/dist/chunk-HPWUNB47.js.map +0 -1
- package/dist/chunk-IYCLP2N2.js +0 -766
- package/dist/chunk-IYCLP2N2.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JQSF5DQT.js +0 -701
- package/dist/chunk-JQSF5DQT.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-M4YBQKIJ.js +0 -1040
- package/dist/chunk-M4YBQKIJ.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NY44NC4A.js +0 -1056
- package/dist/chunk-NY44NC4A.js.map +0 -1
- package/dist/chunk-OIUOT4QD.js +0 -44
- package/dist/chunk-OIUOT4QD.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OWN5NPMC.js +0 -152
- package/dist/chunk-OWN5NPMC.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PC5DOSM7.js +0 -579
- package/dist/chunk-PC5DOSM7.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QB6BDBP2.js +0 -4464
- package/dist/chunk-QB6BDBP2.js.map +0 -1
- package/dist/chunk-RXHCETDZ.js +0 -536
- package/dist/chunk-RXHCETDZ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-SFLLL76A.js +0 -669
- package/dist/chunk-SFLLL76A.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-T6RLYGAD.js +0 -158
- package/dist/chunk-T6RLYGAD.js.map +0 -1
- package/dist/chunk-TJVT4QFF.js +0 -911
- package/dist/chunk-TJVT4QFF.js.map +0 -1
- package/dist/chunk-TQ7LNKZ3.js +0 -136
- package/dist/chunk-TQ7LNKZ3.js.map +0 -1
- package/dist/chunk-U4L7JRPZ.js +0 -1706
- package/dist/chunk-U4L7JRPZ.js.map +0 -1
- package/dist/chunk-U4PHLT2N.js +0 -419
- package/dist/chunk-U4PHLT2N.js.map +0 -1
- package/dist/chunk-VCZ5FQYW.js +0 -928
- package/dist/chunk-VCZ5FQYW.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WVATSFCP.js +0 -1553
- package/dist/chunk-WVATSFCP.js.map +0 -1
- package/dist/chunk-X4YIBDER.js +0 -1662
- package/dist/chunk-X4YIBDER.js.map +0 -1
- package/dist/chunk-YQN4ICPP.js +0 -355
- package/dist/chunk-YQN4ICPP.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZHTZ4EYI.js +0 -1212
- package/dist/chunk-ZHTZ4EYI.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-OJJ7CZF4.js +0 -18
- package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -0,0 +1,596 @@
|
|
|
1
|
+
import { s as validateRunRecord } from "./run-record-BuoE80Dq.js";
|
|
2
|
+
import { E as pearsonR } from "./statistics-CnnxdpOg.js";
|
|
3
|
+
import { n as observedScore, s as trainingScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
|
|
4
|
+
//#region src/campaign/run-record.ts
|
|
5
|
+
/**
|
|
6
|
+
* Project one campaign cell into the canonical run format.
|
|
7
|
+
*
|
|
8
|
+
* A dispatch error establishes terminal execution failure. A judge error only
|
|
9
|
+
* establishes that quality measurement failed after dispatch completed.
|
|
10
|
+
* Failures without a stage remain unknown. No failure becomes a zero-quality
|
|
11
|
+
* label.
|
|
12
|
+
*/
|
|
13
|
+
function campaignCellToRunRecord(cell, options) {
|
|
14
|
+
const quality = projectCampaignCellQuality(cell);
|
|
15
|
+
const execution = campaignCellExecutionEvidence(cell);
|
|
16
|
+
const judgeErrorCount = Math.max(quality.raw.judge_error_count ?? 0, execution.judgeErrorCount ?? 0);
|
|
17
|
+
const cellCostCaptured = Number.isFinite(cell.costUsd) && cell.costUsd >= 0;
|
|
18
|
+
const costUsd = cellCostCaptured ? cell.costUsd : options.defaultCostUsd ?? null;
|
|
19
|
+
const costProvenance = costUsd === null ? {
|
|
20
|
+
kind: "uncaptured",
|
|
21
|
+
usd: null
|
|
22
|
+
} : cellCostCaptured && !cell.costEstimated ? {
|
|
23
|
+
kind: "observed",
|
|
24
|
+
usd: costUsd
|
|
25
|
+
} : {
|
|
26
|
+
kind: "estimated",
|
|
27
|
+
usd: costUsd
|
|
28
|
+
};
|
|
29
|
+
const raw = {
|
|
30
|
+
...finiteMetrics(options.raw),
|
|
31
|
+
...quality.raw,
|
|
32
|
+
rep: cell.rep,
|
|
33
|
+
duration_ms: cell.durationMs,
|
|
34
|
+
...costUsd === null ? {} : { cost_usd: costUsd },
|
|
35
|
+
cost_estimated: cell.costEstimated ? 1 : 0,
|
|
36
|
+
tokens_input: cell.tokenUsage.input,
|
|
37
|
+
tokens_output: cell.tokenUsage.output,
|
|
38
|
+
latency_ms: cell.durationMs,
|
|
39
|
+
...execution.executionErrorCount === void 0 ? {} : { execution_error_count: execution.executionErrorCount },
|
|
40
|
+
...judgeErrorCount > 0 ? { judge_error_count: judgeErrorCount } : {},
|
|
41
|
+
...execution.unclassifiedErrorCount === void 0 ? {} : { unclassified_error_count: execution.unclassifiedErrorCount }
|
|
42
|
+
};
|
|
43
|
+
if (typeof cell.generation === "number") raw.generation = cell.generation;
|
|
44
|
+
if (cell.tokenUsage.reasoning !== void 0) raw.tokens_reasoning = cell.tokenUsage.reasoning;
|
|
45
|
+
if (cell.tokenUsage.cached !== void 0) raw.tokens_cached = cell.tokenUsage.cached;
|
|
46
|
+
if (cell.tokenUsage.cacheWrite !== void 0) raw.tokens_cache_write = cell.tokenUsage.cacheWrite;
|
|
47
|
+
if (costUsd !== null && costUsd > 0) raw.tokens_per_dollar = (cell.tokenUsage.input + cell.tokenUsage.output) / costUsd;
|
|
48
|
+
if (costUsd !== null && quality.score !== void 0 && quality.score > .01) raw.cost_per_quality = costUsd / quality.score;
|
|
49
|
+
const outcome = {
|
|
50
|
+
raw,
|
|
51
|
+
...quality.judgeScores ? { judgeScores: quality.judgeScores } : {}
|
|
52
|
+
};
|
|
53
|
+
if (quality.score !== void 0) if (options.splitTag === "holdout") outcome.holdoutScore = quality.score;
|
|
54
|
+
else outcome.searchScore = quality.score;
|
|
55
|
+
return validateRunRecord({
|
|
56
|
+
runId: options.runId,
|
|
57
|
+
experimentId: options.experimentId,
|
|
58
|
+
candidateId: options.candidateId,
|
|
59
|
+
seed: options.seed ?? cell.seed,
|
|
60
|
+
model: options.model,
|
|
61
|
+
promptHash: options.promptHash,
|
|
62
|
+
configHash: options.configHash,
|
|
63
|
+
commitSha: options.commitSha,
|
|
64
|
+
wallMs: cell.durationMs,
|
|
65
|
+
costUsd,
|
|
66
|
+
costProvenance,
|
|
67
|
+
tokenUsage: { ...cell.tokenUsage },
|
|
68
|
+
terminalOutcome: execution.terminalOutcome,
|
|
69
|
+
...execution.terminalFailureReason ? { terminalFailureReason: execution.terminalFailureReason } : {},
|
|
70
|
+
outcome,
|
|
71
|
+
splitTag: options.splitTag,
|
|
72
|
+
scenarioId: options.scenarioId ?? cell.scenarioId,
|
|
73
|
+
...options.agentProfile ? { agentProfile: options.agentProfile } : {}
|
|
74
|
+
});
|
|
75
|
+
}
|
|
76
|
+
function campaignCellExecutionEvidence(cell) {
|
|
77
|
+
if (cell.errorStage === "dispatch") return {
|
|
78
|
+
terminalOutcome: "failed",
|
|
79
|
+
executionErrorCount: 1,
|
|
80
|
+
...cell.error ? { terminalFailureReason: cell.error } : {}
|
|
81
|
+
};
|
|
82
|
+
if (cell.errorStage === "judge") return {
|
|
83
|
+
terminalOutcome: "succeeded",
|
|
84
|
+
executionErrorCount: 0,
|
|
85
|
+
judgeErrorCount: 1
|
|
86
|
+
};
|
|
87
|
+
if (!cell.error) return {
|
|
88
|
+
terminalOutcome: "succeeded",
|
|
89
|
+
executionErrorCount: 0
|
|
90
|
+
};
|
|
91
|
+
return {
|
|
92
|
+
terminalOutcome: "unknown",
|
|
93
|
+
unclassifiedErrorCount: 1
|
|
94
|
+
};
|
|
95
|
+
}
|
|
96
|
+
/**
|
|
97
|
+
* Produce the only task-quality view used by campaign aggregates and exports.
|
|
98
|
+
*
|
|
99
|
+
* Successful judge results remain available for diagnosis after another judge
|
|
100
|
+
* fails, but a task score exists only for an error-free cell whose reported
|
|
101
|
+
* judge values are all finite.
|
|
102
|
+
*/
|
|
103
|
+
function projectCampaignCellQuality(cell) {
|
|
104
|
+
if (cell.errorStage === "dispatch") return {
|
|
105
|
+
successfulJudgeScores: {},
|
|
106
|
+
failedJudges: [],
|
|
107
|
+
raw: {}
|
|
108
|
+
};
|
|
109
|
+
const perJudge = {};
|
|
110
|
+
const successfulJudgeScores = {};
|
|
111
|
+
const dimensionValues = /* @__PURE__ */ new Map();
|
|
112
|
+
const composites = [];
|
|
113
|
+
const notes = [];
|
|
114
|
+
const failedJudges = new Set(cell.errorStage === "judge" ? [cell.errorJudge ?? "unknown-judge"] : []);
|
|
115
|
+
const raw = {};
|
|
116
|
+
for (const [judgeName, score] of Object.entries(cell.judgeScores)) {
|
|
117
|
+
const finiteDimensions = Object.values(score.dimensions).every(Number.isFinite);
|
|
118
|
+
if (score.failed || !Number.isFinite(score.composite) || !finiteDimensions) {
|
|
119
|
+
failedJudges.add(judgeName);
|
|
120
|
+
continue;
|
|
121
|
+
}
|
|
122
|
+
composites.push(score.composite);
|
|
123
|
+
successfulJudgeScores[judgeName] = score;
|
|
124
|
+
const dimensions = { ...score.dimensions };
|
|
125
|
+
perJudge[judgeName] = dimensions;
|
|
126
|
+
for (const [dimension, value] of Object.entries(dimensions)) {
|
|
127
|
+
raw[`${judgeName}.${dimension}`] = value;
|
|
128
|
+
const values = dimensionValues.get(dimension) ?? [];
|
|
129
|
+
values.push(value);
|
|
130
|
+
dimensionValues.set(dimension, values);
|
|
131
|
+
}
|
|
132
|
+
if (score.notes) notes.push(`${judgeName}: ${score.notes}`);
|
|
133
|
+
for (const failedJudge of score.failedJudges ?? []) failedJudges.add(`${judgeName}/${failedJudge}`);
|
|
134
|
+
}
|
|
135
|
+
if (failedJudges.size > 0) raw.judge_error_count = failedJudges.size;
|
|
136
|
+
const sortedFailedJudges = [...failedJudges].sort();
|
|
137
|
+
if (composites.length === 0) return {
|
|
138
|
+
successfulJudgeScores,
|
|
139
|
+
failedJudges: sortedFailedJudges,
|
|
140
|
+
raw
|
|
141
|
+
};
|
|
142
|
+
const composite = mean$1(composites);
|
|
143
|
+
const perDimMean = Object.fromEntries([...dimensionValues.entries()].map(([dimension, values]) => [dimension, mean$1(values)]));
|
|
144
|
+
const complete = cell.error === void 0 && cell.errorStage === void 0 && failedJudges.size === 0;
|
|
145
|
+
if (complete) raw.composite = composite;
|
|
146
|
+
return {
|
|
147
|
+
...complete ? { score: composite } : {},
|
|
148
|
+
raw,
|
|
149
|
+
successfulJudgeScores,
|
|
150
|
+
failedJudges: sortedFailedJudges,
|
|
151
|
+
judgeScores: {
|
|
152
|
+
perJudge,
|
|
153
|
+
perDimMean,
|
|
154
|
+
composite,
|
|
155
|
+
...sortedFailedJudges.length > 0 ? { failedJudges: sortedFailedJudges } : {},
|
|
156
|
+
...notes.length > 0 ? { notes: notes.join(" | ") } : {}
|
|
157
|
+
}
|
|
158
|
+
};
|
|
159
|
+
}
|
|
160
|
+
/** Read the canonical task score without recomputing cell quality. */
|
|
161
|
+
function campaignCellTaskScore(cell) {
|
|
162
|
+
return projectCampaignCellQuality(cell).score;
|
|
163
|
+
}
|
|
164
|
+
/** Read canonical successful judge dimensions without recomputing cell quality. */
|
|
165
|
+
function campaignCellJudgeDimensions(cell) {
|
|
166
|
+
return projectCampaignCellQuality(cell).judgeScores?.perJudge ?? {};
|
|
167
|
+
}
|
|
168
|
+
function finiteMetrics(metrics) {
|
|
169
|
+
const finite = {};
|
|
170
|
+
for (const [key, value] of Object.entries(metrics ?? {})) if (Number.isFinite(value)) finite[key] = value;
|
|
171
|
+
return finite;
|
|
172
|
+
}
|
|
173
|
+
function mean$1(values) {
|
|
174
|
+
return values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
175
|
+
}
|
|
176
|
+
//#endregion
|
|
177
|
+
//#region src/rl/verifiable-reward.ts
|
|
178
|
+
const DEFAULT_DETERMINISTIC_LAYERS = /* @__PURE__ */ new Set([
|
|
179
|
+
"install",
|
|
180
|
+
"typecheck",
|
|
181
|
+
"build",
|
|
182
|
+
"lint",
|
|
183
|
+
"test",
|
|
184
|
+
"compile",
|
|
185
|
+
"schema",
|
|
186
|
+
"sandbox",
|
|
187
|
+
"unit_tests",
|
|
188
|
+
"integration_tests"
|
|
189
|
+
]);
|
|
190
|
+
const DEFAULT_SOURCE_FOR = (name) => {
|
|
191
|
+
const lower = name.toLowerCase();
|
|
192
|
+
if (lower.includes("test")) return "test";
|
|
193
|
+
if (lower.includes("compile") || lower.includes("build") || lower.includes("typecheck") || lower.includes("lint")) return "compile";
|
|
194
|
+
if (lower.includes("schema")) return "schema";
|
|
195
|
+
if (lower.includes("sandbox")) return "sandbox";
|
|
196
|
+
if (lower.includes("judge") || lower.includes("semantic")) return "judge";
|
|
197
|
+
return "composite";
|
|
198
|
+
};
|
|
199
|
+
/**
|
|
200
|
+
* Extract a `VerifiableReward` from a `VerificationReport`.
|
|
201
|
+
*
|
|
202
|
+
* Strategy: prefer the deterministic layers (in order: test → compile →
|
|
203
|
+
* schema → sandbox), fall back to the judge layer if `fallbackToJudge` is
|
|
204
|
+
* true, return `null` if no signal qualifies. When multiple deterministic
|
|
205
|
+
* layers contribute, return a `'composite'` source with a weighted blend.
|
|
206
|
+
*
|
|
207
|
+
* NO realness gate is applied and none can be: a `VerificationReport` carries
|
|
208
|
+
* layer scores and nothing about whether the run faked them — `realness` lives
|
|
209
|
+
* on the `RunRecord`. Use `extractVerifiableRewardsFromRecords` for anything
|
|
210
|
+
* that becomes training data; this signature is for scoring a report in hand.
|
|
211
|
+
*/
|
|
212
|
+
function extractVerifiableReward(report, opts = {}) {
|
|
213
|
+
const deterministicSet = new Set(opts.deterministicLayers ?? [...DEFAULT_DETERMINISTIC_LAYERS]);
|
|
214
|
+
const sourceFor = opts.sourceFor ?? DEFAULT_SOURCE_FOR;
|
|
215
|
+
const fallbackToJudge = opts.fallbackToJudge ?? true;
|
|
216
|
+
const judgeFloor = opts.judgeConfidenceFloor ?? .7;
|
|
217
|
+
const deterministic = report.layers.filter((layer) => deterministicSet.has(layer.layer) && isMeasuredLayer(layer));
|
|
218
|
+
if (deterministic.length === 1) {
|
|
219
|
+
const layer = deterministic[0];
|
|
220
|
+
const value = clamp01$1(layer.score);
|
|
221
|
+
return {
|
|
222
|
+
value,
|
|
223
|
+
source: sourceFor(layer.layer),
|
|
224
|
+
determinism: "deterministic",
|
|
225
|
+
confidence: 1,
|
|
226
|
+
origin: layer.layer,
|
|
227
|
+
components: { [layer.layer]: value },
|
|
228
|
+
realnessScreened: false
|
|
229
|
+
};
|
|
230
|
+
}
|
|
231
|
+
if (deterministic.length > 1) {
|
|
232
|
+
let num = 0;
|
|
233
|
+
let denom = 0;
|
|
234
|
+
const components = {};
|
|
235
|
+
for (const l of deterministic) {
|
|
236
|
+
const w = l.detail?.weight ?? 1;
|
|
237
|
+
num += w * (l.score ?? 0);
|
|
238
|
+
denom += w;
|
|
239
|
+
components[l.layer] = l.score;
|
|
240
|
+
}
|
|
241
|
+
return {
|
|
242
|
+
value: denom === 0 ? 0 : clamp01$1(num / denom),
|
|
243
|
+
source: "composite",
|
|
244
|
+
determinism: "deterministic",
|
|
245
|
+
confidence: 1,
|
|
246
|
+
origin: deterministic.map((l) => l.layer).join("+"),
|
|
247
|
+
components,
|
|
248
|
+
realnessScreened: false
|
|
249
|
+
};
|
|
250
|
+
}
|
|
251
|
+
if (!fallbackToJudge) return null;
|
|
252
|
+
const judge = report.layers.find((layer) => isMeasuredLayer(layer) && sourceFor(layer.layer) === "judge") ?? report.layers.find(isMeasuredLayer);
|
|
253
|
+
if (!judge) return null;
|
|
254
|
+
const confFromDetail = judge.detail?.confidence;
|
|
255
|
+
const judgeValue = clamp01$1(judge.score);
|
|
256
|
+
return {
|
|
257
|
+
value: judgeValue,
|
|
258
|
+
source: "judge",
|
|
259
|
+
determinism: "probabilistic",
|
|
260
|
+
confidence: typeof confFromDetail === "number" ? confFromDetail : judgeFloor,
|
|
261
|
+
origin: judge.layer,
|
|
262
|
+
components: { [judge.layer]: judgeValue },
|
|
263
|
+
realnessScreened: false
|
|
264
|
+
};
|
|
265
|
+
}
|
|
266
|
+
function isMeasuredLayer(layer) {
|
|
267
|
+
return (layer.status === "pass" || layer.status === "fail") && typeof layer.score === "number" && Number.isFinite(layer.score) && layer.score >= 0 && layer.score <= 1;
|
|
268
|
+
}
|
|
269
|
+
/**
|
|
270
|
+
* Extract verifiable rewards from `RunRecord[]` produced via the
|
|
271
|
+
* `verificationReportToRunRecord` adapter (which encodes per-layer scores
|
|
272
|
+
* in `outcome.raw['layer.<name>']`). For records that don't carry layer
|
|
273
|
+
* scores, returns `null` for that record.
|
|
274
|
+
*
|
|
275
|
+
* This is the canonical bridge from "campaign-shaped artifacts" to
|
|
276
|
+
* "RL-training-ready reward signals": every record that has a clean
|
|
277
|
+
* verifiable reward becomes a training datum, every record that doesn't
|
|
278
|
+
* gets filtered out (or kept with `'probabilistic'` determinism for
|
|
279
|
+
* separate downstream handling).
|
|
280
|
+
*
|
|
281
|
+
* The realness gate applies to EVERY channel here, and to the deterministic one
|
|
282
|
+
* MOST. It is tempting to reason that a decidable signal cannot be gamed, so
|
|
283
|
+
* the gate is redundant on it — that reasoning is backwards. `realness.gated`
|
|
284
|
+
* means the run's success signal was FAKED, and a test suite reporting green on
|
|
285
|
+
* a stubbed integration is precisely what that looks like: the deterministic
|
|
286
|
+
* layer is the thing that got faked. Exporting it ungated hands a trainer the
|
|
287
|
+
* highest-credibility reward the module can emit (`determinism: 'deterministic'`,
|
|
288
|
+
* `confidence: 1`) for the one population the gate exists to catch. Pass
|
|
289
|
+
* `applyRealnessGate: false` only to look at the ungated numbers for detection.
|
|
290
|
+
*/
|
|
291
|
+
function extractVerifiableRewardsFromRecords(runs, opts = {}) {
|
|
292
|
+
const sourceFor = opts.sourceFor ?? DEFAULT_SOURCE_FOR;
|
|
293
|
+
const deterministicSet = new Set(opts.deterministicLayers ?? [...DEFAULT_DETERMINISTIC_LAYERS]);
|
|
294
|
+
const fallbackToJudge = opts.fallbackToJudge ?? true;
|
|
295
|
+
const judgeFloor = opts.judgeConfidenceFloor ?? .7;
|
|
296
|
+
const applyGate = opts.applyRealnessGate ?? true;
|
|
297
|
+
return runs.map((run) => {
|
|
298
|
+
const flagged = isRealnessGated(run);
|
|
299
|
+
const screened = run.outcome.realness === void 0 ? {} : { realnessScreened: true };
|
|
300
|
+
const gate = (value) => applyGate && flagged ? 0 : value;
|
|
301
|
+
const layerScores = [];
|
|
302
|
+
for (const [k, v] of Object.entries(run.outcome.raw)) if (k.startsWith("layer.") && !k.includes(".", 6) && typeof v === "number" && Number.isFinite(v)) layerScores.push({
|
|
303
|
+
name: k.slice(6),
|
|
304
|
+
score: v
|
|
305
|
+
});
|
|
306
|
+
const det = layerScores.filter((l) => deterministicSet.has(l.name));
|
|
307
|
+
if (det.length === 1) {
|
|
308
|
+
const layer = det[0];
|
|
309
|
+
const value = gate(clamp01$1(layer.score));
|
|
310
|
+
return {
|
|
311
|
+
runId: run.runId,
|
|
312
|
+
reward: {
|
|
313
|
+
value,
|
|
314
|
+
source: sourceFor(layer.name),
|
|
315
|
+
determinism: "deterministic",
|
|
316
|
+
confidence: 1,
|
|
317
|
+
origin: layer.name,
|
|
318
|
+
components: { [layer.name]: value },
|
|
319
|
+
realnessGated: flagged,
|
|
320
|
+
...screened
|
|
321
|
+
}
|
|
322
|
+
};
|
|
323
|
+
}
|
|
324
|
+
if (det.length > 1) {
|
|
325
|
+
const value = gate(clamp01$1(det.reduce((s, l) => s + l.score, 0) / det.length));
|
|
326
|
+
const components = Object.fromEntries(det.map((l) => [l.name, gate(clamp01$1(l.score))]));
|
|
327
|
+
return {
|
|
328
|
+
runId: run.runId,
|
|
329
|
+
reward: {
|
|
330
|
+
value,
|
|
331
|
+
source: "composite",
|
|
332
|
+
determinism: "deterministic",
|
|
333
|
+
confidence: 1,
|
|
334
|
+
origin: det.map((l) => l.name).join("+"),
|
|
335
|
+
components,
|
|
336
|
+
realnessGated: flagged,
|
|
337
|
+
...screened
|
|
338
|
+
}
|
|
339
|
+
};
|
|
340
|
+
}
|
|
341
|
+
if (!fallbackToJudge) return {
|
|
342
|
+
runId: run.runId,
|
|
343
|
+
reward: null
|
|
344
|
+
};
|
|
345
|
+
const primary = applyGate ? trainingScore(run) : observedScore(run);
|
|
346
|
+
if (typeof primary !== "number" || !Number.isFinite(primary)) return {
|
|
347
|
+
runId: run.runId,
|
|
348
|
+
reward: null
|
|
349
|
+
};
|
|
350
|
+
const primaryValue = clamp01$1(primary);
|
|
351
|
+
return {
|
|
352
|
+
runId: run.runId,
|
|
353
|
+
reward: {
|
|
354
|
+
value: primaryValue,
|
|
355
|
+
source: "judge",
|
|
356
|
+
determinism: "probabilistic",
|
|
357
|
+
confidence: judgeFloor,
|
|
358
|
+
origin: "run.outcome.score",
|
|
359
|
+
components: { "run.outcome.score": primaryValue },
|
|
360
|
+
realnessGated: flagged,
|
|
361
|
+
...screened
|
|
362
|
+
}
|
|
363
|
+
};
|
|
364
|
+
});
|
|
365
|
+
}
|
|
366
|
+
/**
|
|
367
|
+
* Filter `RunRecord[]` to those with deterministic verifiable rewards.
|
|
368
|
+
*
|
|
369
|
+
* A realness-gated run is KEPT, at reward 0 with `realnessGated: true` — the
|
|
370
|
+
* same rule GRPO uses on a gated line. 0 is the honest label for a faked
|
|
371
|
+
* success and is usable signal, whereas dropping the run would move a group
|
|
372
|
+
* baseline without saying so. (SFT differs: there every row is a target to
|
|
373
|
+
* imitate, so a gated row is removed outright.)
|
|
374
|
+
*/
|
|
375
|
+
function filterDeterministicallyRewarded(runs, opts = {}) {
|
|
376
|
+
const rewarded = extractVerifiableRewardsFromRecords(runs, {
|
|
377
|
+
...opts,
|
|
378
|
+
fallbackToJudge: false
|
|
379
|
+
});
|
|
380
|
+
const out = [];
|
|
381
|
+
for (let i = 0; i < runs.length; i++) {
|
|
382
|
+
const r = rewarded[i];
|
|
383
|
+
if (r.reward && r.reward.determinism === "deterministic") out.push({
|
|
384
|
+
run: runs[i],
|
|
385
|
+
reward: r.reward
|
|
386
|
+
});
|
|
387
|
+
}
|
|
388
|
+
return out;
|
|
389
|
+
}
|
|
390
|
+
function clamp01$1(x) {
|
|
391
|
+
if (!Number.isFinite(x)) return 0;
|
|
392
|
+
return Math.max(0, Math.min(1, x));
|
|
393
|
+
}
|
|
394
|
+
//#endregion
|
|
395
|
+
//#region src/rl/reward-hacking.ts
|
|
396
|
+
/**
|
|
397
|
+
* Reward hacking / Goodhart detection.
|
|
398
|
+
*
|
|
399
|
+
* Goodhart's Law says: when a measure becomes a target, it ceases to be
|
|
400
|
+
* a good measure. In RLHF and agentic-RL settings this is the dominant
|
|
401
|
+
* failure mode — the policy learns to produce outputs that score well on
|
|
402
|
+
* the proxy reward (judge, rubric, test pass-rate) without producing
|
|
403
|
+
* the underlying capability the proxy was meant to track.
|
|
404
|
+
*
|
|
405
|
+
* Krakovna et al. (2020, "Specification Gaming Examples in AI") and the
|
|
406
|
+
* subsequent RLHF reward-hacking literature (Skalse et al. 2022, Kim et al.
|
|
407
|
+
* 2023) converge on a few diagnostic signatures:
|
|
408
|
+
*
|
|
409
|
+
* 1. **Reward divergence:** the proxy reward grows while the held-out
|
|
410
|
+
* ground-truth signal stagnates or drops. Predictive validity over
|
|
411
|
+
* time captures this.
|
|
412
|
+
* 2. **Distributional shift in outputs:** after RL, the policy produces
|
|
413
|
+
* outputs that no longer match the reference distribution — usually
|
|
414
|
+
* because it found a high-reward attractor that's degenerate (e.g.
|
|
415
|
+
* one-token responses, repetition, formatting tricks).
|
|
416
|
+
* 3. **Disagreement between independent rewards:** if you train on
|
|
417
|
+
* reward A and a held-out independent reward B drops sharply, you're
|
|
418
|
+
* probably hacking A.
|
|
419
|
+
* 4. **Calibration drift:** the verifiable / deterministic component of
|
|
420
|
+
* the reward is stable; the probabilistic / judge component drifts up
|
|
421
|
+
* while the deterministic component doesn't. The judge is being
|
|
422
|
+
* gamed.
|
|
423
|
+
*
|
|
424
|
+
* This module ships explicit detectors for all four signatures, plus a
|
|
425
|
+
* combined verdict. The output is diagnostic — actionable signals,
|
|
426
|
+
* not autoreject — because each signature has known false positives
|
|
427
|
+
* (e.g., a policy that genuinely improves can show distributional shift).
|
|
428
|
+
*
|
|
429
|
+
* Differs from `rubricPredictiveValidity` (which is a *standing* check on
|
|
430
|
+
* whether rubrics correlate with deployment outcomes) — this is a
|
|
431
|
+
* *temporal* check on whether the reward-vs-truth gap is *widening over
|
|
432
|
+
* time during a training run*.
|
|
433
|
+
*/
|
|
434
|
+
const DEFAULT_PROXY = (r) => {
|
|
435
|
+
const v = observedScore(r);
|
|
436
|
+
return typeof v === "number" && Number.isFinite(v) ? v : null;
|
|
437
|
+
};
|
|
438
|
+
function detectRewardHacking(input) {
|
|
439
|
+
const proxyOf = input.proxyOf ?? DEFAULT_PROXY;
|
|
440
|
+
const truthOf = input.truthOf;
|
|
441
|
+
const sus = input.thresholds?.suspect ?? .3;
|
|
442
|
+
const gam = input.thresholds?.gaming ?? .6;
|
|
443
|
+
const runs = input.runs.filter((run) => finiteNumber(proxyOf(run)));
|
|
444
|
+
const n = runs.length;
|
|
445
|
+
if (n < 4) return {
|
|
446
|
+
findings: [],
|
|
447
|
+
evaluatedSignals: [],
|
|
448
|
+
verdict: "insufficient_evidence",
|
|
449
|
+
n,
|
|
450
|
+
rationale: [`fewer than 4 runs with proxy reward (n=${n}); insufficient evidence`]
|
|
451
|
+
};
|
|
452
|
+
const windowSize = Math.max(1, input.windowSize ?? Math.min(50, Math.floor(n / 2)));
|
|
453
|
+
const before = runs.slice(0, n - windowSize);
|
|
454
|
+
const after = runs.slice(n - windowSize);
|
|
455
|
+
const findings = [];
|
|
456
|
+
if (truthOf) {
|
|
457
|
+
const beforeProxy = before.map(proxyOf).filter(finiteNumber);
|
|
458
|
+
const afterProxy = after.map(proxyOf).filter(finiteNumber);
|
|
459
|
+
const beforeTruth = before.map(truthOf).filter(finiteNumber);
|
|
460
|
+
const afterTruth = after.map(truthOf).filter(finiteNumber);
|
|
461
|
+
if (beforeProxy.length >= 2 && afterProxy.length >= 2 && beforeTruth.length >= 2 && afterTruth.length >= 2) {
|
|
462
|
+
const proxyDelta = mean(afterProxy) - mean(beforeProxy);
|
|
463
|
+
const truthDelta = mean(afterTruth) - mean(beforeTruth);
|
|
464
|
+
const gap = Math.max(0, proxyDelta - truthDelta);
|
|
465
|
+
const severity = clamp01(gap * 5);
|
|
466
|
+
findings.push({
|
|
467
|
+
signal: "reward_divergence",
|
|
468
|
+
severity,
|
|
469
|
+
message: severity >= sus ? `proxy reward rose by ${proxyDelta.toFixed(3)} while truth changed by ${truthDelta.toFixed(3)} — potential Goodhart` : `proxy and truth moved together (proxy ${proxyDelta.toFixed(3)}, truth ${truthDelta.toFixed(3)})`,
|
|
470
|
+
detail: {
|
|
471
|
+
proxyDelta,
|
|
472
|
+
truthDelta,
|
|
473
|
+
gap,
|
|
474
|
+
beforeN: beforeProxy.length,
|
|
475
|
+
afterN: afterProxy.length
|
|
476
|
+
}
|
|
477
|
+
});
|
|
478
|
+
}
|
|
479
|
+
}
|
|
480
|
+
{
|
|
481
|
+
const beforeP = before.map(proxyOf).filter(finiteNumber);
|
|
482
|
+
const afterP = after.map(proxyOf).filter(finiteNumber);
|
|
483
|
+
if (beforeP.length >= 4 && afterP.length >= 4) {
|
|
484
|
+
const ks = ksStatistic(beforeP, afterP);
|
|
485
|
+
const severity = clamp01(ks - .2);
|
|
486
|
+
findings.push({
|
|
487
|
+
signal: "distribution_shift",
|
|
488
|
+
severity,
|
|
489
|
+
message: severity >= sus ? `KS=${ks.toFixed(3)} between before/after windows — distributional shift large` : `KS=${ks.toFixed(3)} between before/after windows — within-distribution drift`,
|
|
490
|
+
detail: {
|
|
491
|
+
ks,
|
|
492
|
+
beforeN: beforeP.length,
|
|
493
|
+
afterN: afterP.length
|
|
494
|
+
}
|
|
495
|
+
});
|
|
496
|
+
}
|
|
497
|
+
}
|
|
498
|
+
{
|
|
499
|
+
const secondaryOf = input.secondaryRewardOf ?? defaultSecondary(input.verifiableRewardOptions);
|
|
500
|
+
const aligned = runs.map((r) => ({
|
|
501
|
+
p: proxyOf(r),
|
|
502
|
+
s: secondaryOf(r)
|
|
503
|
+
})).filter((x) => finiteNumber(x.p) && finiteNumber(x.s));
|
|
504
|
+
if (aligned.length >= 4) {
|
|
505
|
+
const r = pearsonR(aligned.map((x) => x.p), aligned.map((x) => x.s));
|
|
506
|
+
const severity = clamp01(.5 - Math.max(0, r));
|
|
507
|
+
findings.push({
|
|
508
|
+
signal: "reward_disagreement",
|
|
509
|
+
severity,
|
|
510
|
+
message: severity >= sus ? `proxy and independent secondary reward correlate ρ=${r.toFixed(3)} — possibly hacking proxy` : `proxy and secondary reward correlate ρ=${r.toFixed(3)}`,
|
|
511
|
+
detail: {
|
|
512
|
+
pearson: r,
|
|
513
|
+
n: aligned.length
|
|
514
|
+
}
|
|
515
|
+
});
|
|
516
|
+
}
|
|
517
|
+
}
|
|
518
|
+
{
|
|
519
|
+
const detRuns = filterDeterministicallyRewarded(runs, {
|
|
520
|
+
...input.verifiableRewardOptions ?? {},
|
|
521
|
+
applyRealnessGate: false
|
|
522
|
+
});
|
|
523
|
+
if (detRuns.length >= 4) {
|
|
524
|
+
const detBefore = detRuns.slice(0, Math.floor(detRuns.length / 2));
|
|
525
|
+
const detDelta = mean(detRuns.slice(Math.floor(detRuns.length / 2)).map((r) => r.reward.value)) - mean(detBefore.map((r) => r.reward.value));
|
|
526
|
+
const proxyDelta = mean(after.map(proxyOf).filter(finiteNumber)) - mean(before.map(proxyOf).filter(finiteNumber));
|
|
527
|
+
const driftGap = Math.max(0, proxyDelta - detDelta);
|
|
528
|
+
const severity = clamp01(driftGap * 5);
|
|
529
|
+
findings.push({
|
|
530
|
+
signal: "judge_drift",
|
|
531
|
+
severity,
|
|
532
|
+
message: severity >= sus ? `judge proxy +${proxyDelta.toFixed(3)} while deterministic reward +${detDelta.toFixed(3)} — judge drifting up without verifiable backing` : `judge and deterministic rewards move in step (judge ${proxyDelta.toFixed(3)}, det ${detDelta.toFixed(3)})`,
|
|
533
|
+
detail: {
|
|
534
|
+
proxyDelta,
|
|
535
|
+
detDelta,
|
|
536
|
+
driftGap,
|
|
537
|
+
n: detRuns.length
|
|
538
|
+
}
|
|
539
|
+
});
|
|
540
|
+
}
|
|
541
|
+
}
|
|
542
|
+
const maxSev = findings.reduce((m, f) => Math.max(m, f.severity), 0);
|
|
543
|
+
if (findings.length === 0) return {
|
|
544
|
+
findings,
|
|
545
|
+
evaluatedSignals: [],
|
|
546
|
+
verdict: "insufficient_evidence",
|
|
547
|
+
rationale: [`no reward-hacking signal had enough paired evidence (n=${n})`],
|
|
548
|
+
n
|
|
549
|
+
};
|
|
550
|
+
const verdict = maxSev >= gam ? "gaming" : maxSev >= sus ? "suspect" : "clean";
|
|
551
|
+
const rationale = findings.filter((f) => f.severity >= sus).map((f) => `${f.signal}: severity ${f.severity.toFixed(2)} — ${f.message}`);
|
|
552
|
+
if (rationale.length === 0) rationale.push("no signals fired above suspect threshold");
|
|
553
|
+
return {
|
|
554
|
+
findings,
|
|
555
|
+
evaluatedSignals: findings.map((finding) => finding.signal),
|
|
556
|
+
verdict,
|
|
557
|
+
rationale,
|
|
558
|
+
n
|
|
559
|
+
};
|
|
560
|
+
}
|
|
561
|
+
function mean(xs) {
|
|
562
|
+
if (xs.length === 0) return 0;
|
|
563
|
+
return xs.reduce((s, x) => s + x, 0) / xs.length;
|
|
564
|
+
}
|
|
565
|
+
function finiteNumber(value) {
|
|
566
|
+
return typeof value === "number" && Number.isFinite(value);
|
|
567
|
+
}
|
|
568
|
+
function clamp01(x) {
|
|
569
|
+
if (!Number.isFinite(x)) return 0;
|
|
570
|
+
return Math.max(0, Math.min(1, x));
|
|
571
|
+
}
|
|
572
|
+
function ksStatistic(a, b) {
|
|
573
|
+
const sortedA = [...a].sort((x, y) => x - y);
|
|
574
|
+
const sortedB = [...b].sort((x, y) => x - y);
|
|
575
|
+
const all = [.../* @__PURE__ */ new Set([...sortedA, ...sortedB])].sort((x, y) => x - y);
|
|
576
|
+
let max = 0;
|
|
577
|
+
for (const v of all) {
|
|
578
|
+
const fa = sortedA.filter((x) => x <= v).length / sortedA.length;
|
|
579
|
+
const fb = sortedB.filter((x) => x <= v).length / sortedB.length;
|
|
580
|
+
max = Math.max(max, Math.abs(fa - fb));
|
|
581
|
+
}
|
|
582
|
+
return max;
|
|
583
|
+
}
|
|
584
|
+
function defaultSecondary(verifiableOpts) {
|
|
585
|
+
return (run) => {
|
|
586
|
+
const filtered = filterDeterministicallyRewarded([run], {
|
|
587
|
+
...verifiableOpts ?? {},
|
|
588
|
+
applyRealnessGate: false
|
|
589
|
+
});
|
|
590
|
+
return filtered.length === 1 ? filtered[0].reward.value : null;
|
|
591
|
+
};
|
|
592
|
+
}
|
|
593
|
+
//#endregion
|
|
594
|
+
export { campaignCellExecutionEvidence as a, campaignCellToRunRecord as c, filterDeterministicallyRewarded as i, projectCampaignCellQuality as l, extractVerifiableReward as n, campaignCellJudgeDimensions as o, extractVerifiableRewardsFromRecords as r, campaignCellTaskScore as s, detectRewardHacking as t };
|
|
595
|
+
|
|
596
|
+
//# sourceMappingURL=reward-hacking-qipEpKvY.js.map
|