@tangle-network/agent-eval 0.128.2 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +279 -0
- package/README.md +19 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +83 -2932
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -364
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1205
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1710
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -894
- package/dist/benchmarks/index.js +2 -59
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6390
- package/dist/campaign/index.js +3 -212
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -174
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5605
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1937
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -32
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -617
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3776 -15120
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11185 -11191
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -481
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1298
- package/dist/reporting.js +6 -50
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +916 -3596
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2362 -1751
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -1048
- package/dist/rollout/index.js +8 -110
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/run-record-BuoE80Dq.js.map +1 -0
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -849
- package/dist/supervisor-run/index.js +2 -64
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -251
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1174
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +18 -10
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2JX3CFMB.js +0 -695
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-2MKQIFS4.js +0 -183
- package/dist/chunk-2MKQIFS4.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BYT7ELPS.js +0 -1553
- package/dist/chunk-BYT7ELPS.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js +0 -2428
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-DPUHNQLN.js +0 -232
- package/dist/chunk-DPUHNQLN.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js +0 -617
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js +0 -2001
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js +0 -1559
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js +0 -171
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MHELPNRP.js +0 -1212
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js +0 -1040
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js +0 -7633
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js +0 -332
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-P5W7RQKK.js +0 -576
- package/dist/chunk-P5W7RQKK.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js +0 -669
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-S5YLIBFX.js +0 -136
- package/dist/chunk-S5YLIBFX.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TBL77AUT.js +0 -355
- package/dist/chunk-TBL77AUT.js.map +0 -1
- package/dist/chunk-TSN7JT6D.js +0 -1646
- package/dist/chunk-TSN7JT6D.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js +0 -4461
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js +0 -291
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js +0 -163
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js +0 -908
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js +0 -929
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js +0 -695
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js +0 -766
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/chunk-YJBNWCAA.js +0 -1056
- package/dist/chunk-YJBNWCAA.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZUUWPZCV.js +0 -752
- package/dist/chunk-ZUUWPZCV.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,285 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.130.0] - 2026-07-26 - current dependency and build cohort
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- Updated Agent Core to `0.4.24` and Agent Interface to `0.35.0`.
|
|
12
|
+
- Updated TypeScript to `7.0.2` and replaced the unsupported declaration build with `tsdown`.
|
|
13
|
+
- Replaced the TypeScript compiler import in the score derivation source check with `oxc-parser`.
|
|
14
|
+
- Updated GitHub Actions to their current stable major releases.
|
|
15
|
+
|
|
16
|
+
### Fixed
|
|
17
|
+
|
|
18
|
+
- Release checks now validate package metadata and every ESM and bundler type entrypoint before publish.
|
|
19
|
+
- Removed two obsolete declaration build scripts and stale release instructions.
|
|
20
|
+
|
|
21
|
+
## [0.129.0] - 2026-07-25 - provider-neutral chat and canonical rollout training
|
|
22
|
+
|
|
23
|
+
### Changed
|
|
24
|
+
|
|
25
|
+
- **Breaking:** benchmark, driver, executor, judge, completion-checker, tracing, and analyst APIs now accept `ChatClient`.
|
|
26
|
+
- **Breaking:** removed the exported provider SDK type, direct provider SDK dependency, provider-specific retry fields, and custom completion-checker error receipt callback.
|
|
27
|
+
- **Breaking:** `LlmClientOptions.maximumAttempts` replaces the misleading `maxRetries` name, which already represented total attempts.
|
|
28
|
+
- **Breaking:** `toGrpoRows`, `toSftRows`, `extractPreferences`, and `buildRlDataset` accept only `MintedRolloutLine[]`.
|
|
29
|
+
Convert run records once with `mintRolloutRows`.
|
|
30
|
+
- **Breaking:** removed `rolloutReward`, record-input training overloads, record-only reward hooks, duplicate line lookup and preference option types, and the duplicate dataset split map.
|
|
31
|
+
- **Breaking:** removed scalar belief-state and off-policy `qHat` fields; contextual estimates require `qHatChosen` and `vHatTarget` together.
|
|
32
|
+
- **Breaking:** removed `CampaignAggregates.totalCostUsd`, `CostLedgerEntry`, `VerifiableReward.breakdown`, and the fixed-prompt `JudgeFn` factories.
|
|
33
|
+
- **Breaking:** removed the unused `OptimizationProposer` alias; use `SurfaceProposer`.
|
|
34
|
+
- **Breaking:** `CampaignStorage.append` is required; read/write-only storage adapters are no longer accepted.
|
|
35
|
+
- **Breaking:** `RawAnalystFinding` now has one plural `evidence` field.
|
|
36
|
+
Removed the duplicate `CanonicalRawAnalystFinding` names, singular-evidence adapters, the second
|
|
37
|
+
recovery callback, `AnalystRunSummary.cost_usd`, and finding-metadata cost accounting.
|
|
38
|
+
- Paid calls read canonical `ChatResponse.content`, usage, model, duration, and cost.
|
|
39
|
+
- Cost reservations derive provider retries from `ChatClient.maximumAttempts`; capped calls reject clients that do not declare a finite attempt count.
|
|
40
|
+
- `createChatClient({ transport: 'custom' })` adapts external SDKs and transports without importing them into Agent Eval.
|
|
41
|
+
- Updated Agent Core to `0.4.22` and Agent Interface to `0.34.0`.
|
|
42
|
+
- No compatibility aliases, overloads, environment fallbacks, or alternate readers preserve these
|
|
43
|
+
removed fields and functions.
|
|
44
|
+
- Settled cost events accept one current receipt shape, and execution summaries read only
|
|
45
|
+
`outcome.raw.execution_error_count`.
|
|
46
|
+
- Single-run locks accept only structured owner records; plain-PID lock files are rejected.
|
|
47
|
+
|
|
48
|
+
### Fixed
|
|
49
|
+
|
|
50
|
+
- **A run flagged as gamed exported at full positive reward through every RL path.** The realness gate
|
|
51
|
+
(`outcome.realness.gated`) existed in exactly one function, `rolloutReward`, called from exactly one
|
|
52
|
+
place — `mintRolloutRows`. The same derivation, `outcome.holdoutScore ?? outcome.searchScore`, was
|
|
53
|
+
hand-rolled at 20 other sites with no gate. Six of those sites feed exported training data: the GRPO
|
|
54
|
+
default reward and the SFT row metadata (`rl/exporters.ts`), the DPO preference ordering
|
|
55
|
+
(`rl/preferences.ts`), the probabilistic verifiable-reward fallback (`rl/verifiable-reward.ts`), the
|
|
56
|
+
corpus `minScore` filter (`rl/corpus.ts`), and the published datasheet's reward statistics
|
|
57
|
+
(`rl/dataset.ts`). A gamed run therefore trained at its claimed score, and in DPO it became the
|
|
58
|
+
*chosen* side of a pair against its honest sibling. Every one of those six now derives its reward
|
|
59
|
+
through the gate.
|
|
60
|
+
|
|
61
|
+
### Added
|
|
62
|
+
|
|
63
|
+
- `trainingScore`, `trainingReward`, `observedScore`, `isRealnessGated`, and the `ScorePreference`
|
|
64
|
+
type — the score derivation now lives once, in `src/rollout/reward.ts`, behind two names that force
|
|
65
|
+
the caller to state intent. `trainingScore` / `trainingReward` are gated and required for anything a
|
|
66
|
+
trainer or an exported dataset consumes; `observedScore` is raw and documented as unsafe for
|
|
67
|
+
training data. Raw is a legitimate choice — reward-hack detection, scorecards, and curriculum
|
|
68
|
+
allocation need the ungated number, and gating a detector's proxy would make it report "clean" on
|
|
69
|
+
precisely the population being gamed — so the fix names the choice rather than removing it.
|
|
70
|
+
- A regression test (`src/rollout/reward-invariant.test.ts`) with two halves: a source-level check that
|
|
71
|
+
the bare derivation appears nowhere outside `rollout/reward.ts`, and a behavioural check that pushes
|
|
72
|
+
one gated record with a 0.95 score through mint, GRPO, SFT, DPO, verifiable reward, the dataset
|
|
73
|
+
bundle, and the corpus filter, asserting 0 in each. Against the pre-fix tree the source check reports
|
|
74
|
+
21 offending lines and 7 of the 10 tests fail.
|
|
75
|
+
|
|
76
|
+
### Changed
|
|
77
|
+
|
|
78
|
+
- `rolloutReward` was removed.
|
|
79
|
+
Use `trainingReward` for score derivation or `rolloutRewardFields` when producing a rollout outcome.
|
|
80
|
+
- The 14 analysis, reporting, and detection sites that legitimately want the raw number now call
|
|
81
|
+
`observedScore` explicitly. Behaviour is unchanged at all 14, including the two sites that
|
|
82
|
+
deliberately prefer the search split (`rl/active-curriculum.ts`, via the new `ScorePreference`
|
|
83
|
+
argument). (`description-length-gate.ts` reads through `runTaskScore`, which 0.127.0 stripped of
|
|
84
|
+
its obsolete `raw.score` fallback.)
|
|
85
|
+
|
|
86
|
+
### Fixed — second pass (the waist now enforces its own invariant)
|
|
87
|
+
|
|
88
|
+
An adversarial review of the pass above found the hole still open on 13 paths. Its core finding:
|
|
89
|
+
`validateRolloutLine({outcome: {reward: 0.95, realness_gated: true}})` returned **zero errors**. It
|
|
90
|
+
type-checked `reward` and it type-checked `realness_gated`, and never once checked the RELATIONSHIP
|
|
91
|
+
between them. `RolloutLine` was a plain structural interface, so any object literal of that shape WAS
|
|
92
|
+
one — "the input is a rollout line" guaranteed nothing, and the ledger round-tripped a poisoned line
|
|
93
|
+
unchanged.
|
|
94
|
+
|
|
95
|
+
- **The invariant now lives in the validator.** `validateRolloutLine` / `assertRolloutLine` reject
|
|
96
|
+
`reward > 0` together with `realness_gated: true`, with an error that explains the rule rather than
|
|
97
|
+
naming the fields. Because `writeRolloutLedger` and `readRolloutLedger` both assert, a poisoned line
|
|
98
|
+
can neither enter a ledger nor leave one — which closes the published-CLI path (`agent-eval
|
|
99
|
+
rollout-release`) at its entrance: `buildHfDataset` reads through `readRolloutLedger`, so the gated
|
|
100
|
+
line is refused before `verifiers/train.jsonl` and `rft/train.jsonl` are written. The dataset card's
|
|
101
|
+
claim about the flag was a **false claim on a published artifact** until now; it is now enforced, and
|
|
102
|
+
the card states the count of gated lines it shipped.
|
|
103
|
+
- **And in the type system.** `MintedRolloutLine` brands the line with a phantom `unique symbol`
|
|
104
|
+
(nothing at runtime, identical JSON). It is produced only by `mintRolloutRows`, `readRolloutLedger`,
|
|
105
|
+
or an explicit `assertMinted` / `assertMintedLines`. The training exporters now require it:
|
|
106
|
+
`rollout/exporters` (`toSftRows`, `toRewardRows`, `toVerifiersRolloutOutput(s)`, `toRftItem(s)`),
|
|
107
|
+
`rl/exporters` (`toGrpoRows`, `toSftRows`, `PrmLineContext.lines`), `rl/preferences.extractPreferences`,
|
|
108
|
+
and `rl/dataset.buildRlDataset`. Belt and braces on purpose: the brand closes first-party call sites
|
|
109
|
+
at compile time, the validator closes data arriving at runtime.
|
|
110
|
+
- **The regex guard is replaced, not extended.** `src/rollout/score-derivation-guard.ts` walks the
|
|
111
|
+
TypeScript AST of `src/**` and flags every READ of `outcome.holdoutScore` / `outcome.searchScore`
|
|
112
|
+
outside a *counted* allowlist (writes and declarations are untouched). The old line regex caught 2 of
|
|
113
|
+
the 7 re-derivations the review planted; the AST rule catches all 7, and they are kept as a permanent
|
|
114
|
+
fixture in `reward-invariant.test.ts` rather than a one-time demonstration.
|
|
115
|
+
- **`supervisorRunRolloutLines` was a second minting door**, writing `outcome.reward` from the judge
|
|
116
|
+
score and omitting `realness_gated` entirely. It now states the flag explicitly on every supervisor
|
|
117
|
+
and worker row, and its rows are plain `RolloutLine`s — a caller putting them into a training export
|
|
118
|
+
has to run them through `assertMinted` first.
|
|
119
|
+
- **`EvalTraceStore.getBest` ranked few-shot exemplars on the ungated score** while its doc comment
|
|
120
|
+
claimed otherwise. `runScore` is now gated, and `getBest` drops realness-gated runs outright instead
|
|
121
|
+
of ranking them: whatever it returns is pasted into the next agent's prompt as an example to imitate,
|
|
122
|
+
so the SFT rule applies. When every run for a scenario is gated the answer is `null`, not the
|
|
123
|
+
least-bad fake.
|
|
124
|
+
- **`release-confidence.passRate` counted a gamed run as a pass.** Gated runs are now excluded from
|
|
125
|
+
both numerator and denominator, and the count ships beside the rate as `metrics.realnessGatedRuns`.
|
|
126
|
+
`HeldOutGate` gets the same treatment: gated runs are dropped from both sides before pairing, with
|
|
127
|
+
`evidence.realnessGatedRuns` surfacing how many. Never a silent 0 — a shrunken denominator has to say
|
|
128
|
+
by how much.
|
|
129
|
+
- **Ten remaining hand-rolled derivations routed by classification**: `rl/sim-fidelity.ts` (×2, RAW —
|
|
130
|
+
gating a sim-vs-production divergence measure would report the simulator as more faithful precisely
|
|
131
|
+
where it is gamed), `belief-state/code-agent-corpus.ts` (GATED — its output becomes corpus labels),
|
|
132
|
+
`eval-trace-store.ts` (GATED, above), `contract/analyze-runs.ts` (×3, RAW), `summary-report.ts` (×4,
|
|
133
|
+
RAW), `release-confidence.ts` (RAW), `held-out-gate.ts` (RAW, over an already-degated set).
|
|
134
|
+
- **`trainingReward` no longer collapses an unscored record to 0.** It returns `reward: null`, matching
|
|
135
|
+
the schema's own "a labeled gap, never 0" rule; a gated run still returns 0, because that IS a
|
|
136
|
+
verdict. Previously a run nobody graded was indistinguishable from one graded a total failure.
|
|
137
|
+
|
|
138
|
+
### Fixed — the published dataset (`agent-eval rollout-release`)
|
|
139
|
+
|
|
140
|
+
- **The card's gate claim is no longer a sentence; it is a rendering of measured counts.** A README that
|
|
141
|
+
STATES what the build does is a claim about bytes it never reads, and it drifts the moment an exporter
|
|
142
|
+
changes — to whoever downloads the dataset. `buildHfDataset` now exports the rows for every selected
|
|
143
|
+
format, measures the realness-gated rows among them (`measureFormatGate`, matched on `rollout_id`),
|
|
144
|
+
checks the measurement against the declared per-format policy, and only then writes. `buildDatasetCard`
|
|
145
|
+
requires that report, renders it, and **throws** if it disagrees with the lines it describes or with the
|
|
146
|
+
policy. A card that contradicts its own data files cannot be produced without failing the build first.
|
|
147
|
+
The measurement also ships on `BuildSummary.gate` and in the CLI's stdout JSON.
|
|
148
|
+
- **Nothing is written when any config would ship a gated row above reward 0.** Formats used to be
|
|
149
|
+
exported and written one at a time; a build that failed halfway left a poisoned config on disk for
|
|
150
|
+
someone to `--push`. Rows are now computed and gate-checked for every format before the first byte.
|
|
151
|
+
- **The per-format decision is stated once as data**, in `src/rollout/release/gate-report.ts`:
|
|
152
|
+
`sft: 'exclude'` (an SFT row is imitated verbatim — a gamed trajectory must not appear at any weight),
|
|
153
|
+
`verifiers` / `rft` / `raw`: `'zero-and-flag'`. Keeping gated rows in the last three is deliberate: in
|
|
154
|
+
`verifiers` the reward is a signed learning signal, so a gamed trajectory at reward 0 is a correct
|
|
155
|
+
negative, and dropping it would bias the negative population toward honest failures and leave a trainer
|
|
156
|
+
no example of gaming being penalized; `rft` re-samples the completion, so only the prompt and the grader
|
|
157
|
+
reference ship; `raw` is an audit dump, where the gated row is the one an auditor most wants.
|
|
158
|
+
- **Reward 0 is never the only label.** Zeroing without the flag makes a faked success indistinguishable
|
|
159
|
+
from an honest failure — it hides the gamed population instead of disclosing it. `VerifiersRolloutOutput.
|
|
160
|
+
info.realness_gated`, `RftItem.reference.realness_gated`, and `RewardRow.metadata.realness_gated` are new
|
|
161
|
+
and always present, so a consumer can filter the population out or select it for a gaming detector.
|
|
162
|
+
|
|
163
|
+
### Fixed — Harbor ATIF interchange (`src/rollout/interchange/harbor.ts`)
|
|
164
|
+
|
|
165
|
+
- **Export emitted documents that violate ATIF MUST rule 2.** Tool results were folded into the
|
|
166
|
+
`observation` of whichever step happened to PRECEDE them, so an assistant turn that declared no tool
|
|
167
|
+
calls could carry a `source_call_id`, a result could be attached to a step that declared a different
|
|
168
|
+
call, and an unanswered result rode a synthetic `system` step carrying a `source_call_id` a system
|
|
169
|
+
step can never declare. Results now attach only to the step that declared their `tool_call_id`;
|
|
170
|
+
everything else becomes a carrier step whose observation states no call id and escrows it instead.
|
|
171
|
+
Message order is preserved exactly in every case, and `ruleTwoViolations` checks the whole tree in
|
|
172
|
+
the tests.
|
|
173
|
+
- **`logprobs`, `prompt_token_ids`, `completion_token_ids` and per-step `llm_call_count` were adopted
|
|
174
|
+
onto our types but never wired through the interchange.** They were escrowed under
|
|
175
|
+
`extra.tangle.spans`, where no foreign consumer looks, and ATIF's own `step.metrics` /
|
|
176
|
+
`step.llm_call_count` were left empty in both directions — so a Harbor-native file's logprobs were
|
|
177
|
+
read, validated, and dropped. They now travel on the native channel both ways (escrow still wins on
|
|
178
|
+
import for exactness); a step carrying none of them still produces no span.
|
|
179
|
+
- **`session_id` was invocation-scoped.** ATIF's `session_id` is RUN-scoped; export set it to the root
|
|
180
|
+
LINE's `rollout_id`, so two roots of one run got different session ids and foreign tooling grouping
|
|
181
|
+
by session split the run. It is now `run_id`, on every node.
|
|
182
|
+
- **`is_copied_context` (RFC rule 7) was silently dropped on import.** It is now a field on
|
|
183
|
+
`ChatMessage`, validated, carried both ways on ATIF's native step field, and — the part the RFC
|
|
184
|
+
actually mandates — `toSftRows` excludes those turns, dropping the row entirely if nothing else is
|
|
185
|
+
left.
|
|
186
|
+
- **An escrowed split was trusted from any document.** `extra.tangle.task.split: 'search'` in a
|
|
187
|
+
hand-written or third-party file imported as a TRAINABLE split; the escrow key is namespaced, not
|
|
188
|
+
authenticated. Import now forces `holdout` unconditionally, and promotion is an explicit, greppable
|
|
189
|
+
step (`relabelImportedSplit`) so `grep` enumerates every place foreign data was declared trainable.
|
|
190
|
+
- **Round-tripping was not idempotent.** `provenance.gap` accreted one copy of the import note per
|
|
191
|
+
pass, and imported messages were assembled in an order that depended on which optional fields were
|
|
192
|
+
present, so a ledger hashed on serialized bytes saw a diff. The gap is now composed as a
|
|
193
|
+
de-duplicated ordered set and every imported message is built in the canonical schema key order.
|
|
194
|
+
- **The interchange was not root-exported.** `import { toHarborTrajectory } from
|
|
195
|
+
'@tangle-network/agent-eval'` failed — the symbols existed only on the `/rollout` subpath while every
|
|
196
|
+
other rollout symbol was on both.
|
|
197
|
+
- **The reward-absence test was a substring scan.** `expect(serialized).not.toContain('reward')` passed
|
|
198
|
+
only because the fixture happened to have no reward-shaped key in `outcome.metrics`; it says nothing
|
|
199
|
+
about WHERE a match is and false-positives on any metric named e.g. `reward_hack_rate`. It is now a
|
|
200
|
+
structural walk that reports the PATH of every label-shaped key, exempting the escrowed metrics bag
|
|
201
|
+
by path.
|
|
202
|
+
|
|
203
|
+
### Known gap (not fixed here)
|
|
204
|
+
|
|
205
|
+
- `mintRolloutRows` hardcodes `tool_defs: []`, so `agent.tool_definitions` is absent on every minted
|
|
206
|
+
line's ATIF export. This is a capture-side gap, not an interchange one: neither `RunRecord` nor the
|
|
207
|
+
trace-span projection carries a tool schema, so there is nothing for mint to read. Fixing it means
|
|
208
|
+
recording the harness's tool definitions at capture time.
|
|
209
|
+
|
|
210
|
+
### Fixed — unrelated flake encountered on the way
|
|
211
|
+
|
|
212
|
+
- `node:sqlite` is loaded through `createRequire` in `rollout/readers/opencode-sqlite.ts` and its test.
|
|
213
|
+
esbuild and Vite both rewrite an `import()` of a builtin and strip the `node:` prefix, producing a bogus
|
|
214
|
+
`sqlite` package lookup; composing the specifier at runtime did not reliably defeat it, so the failure
|
|
215
|
+
moved between workers whenever a test file was added. A require obtained from `createRequire` is not an
|
|
216
|
+
analyzable module reference in either tool.
|
|
217
|
+
|
|
218
|
+
### Added — second pass
|
|
219
|
+
|
|
220
|
+
- `MintedRolloutLine`, `MintedRolloutOutcome`, `assertMinted`, `assertMintedLines` (rollout barrel +
|
|
221
|
+
root barrel).
|
|
222
|
+
- `observedSplitScore` (the raw score on ONE split, no cross-split fallback — what every split-scoped
|
|
223
|
+
report and promotion gate actually wants) and `scoreOrigin` / `ScoreOrigin` (which split carried the
|
|
224
|
+
score, or that none did — the provenance `reward_source` is built from).
|
|
225
|
+
- `malformedRolloutLine` in `rollout/fixtures` for tests whose subject is the validator itself;
|
|
226
|
+
`fixtureRolloutLine` now validates on every construction and returns a `MintedRolloutLine`.
|
|
227
|
+
- `rollout/release/gate-report`: `FORMAT_GATE_DISPOSITION`, `GateDisposition`, `GateReport`,
|
|
228
|
+
`FormatGateCounts`, `ReleaseRowRef`, `gatedRolloutIds`, `releaseRowRefs`, `measureFormatGate`,
|
|
229
|
+
`assertGateReport` (rollout barrel). `BuildSummary.gate` and `DatasetCardInputs.gate` are new;
|
|
230
|
+
`DatasetCardInputs.gate` is required, so a card cannot be rendered without the measurement.
|
|
231
|
+
|
|
232
|
+
### Fixed — third pass (one canonical training input)
|
|
233
|
+
|
|
234
|
+
Trainer-facing APIs now accept canonical minted lines only.
|
|
235
|
+
Artifacts that carry run ids without embedded reward state still require explicit line context.
|
|
236
|
+
|
|
237
|
+
- **Record-input preference and trainer overloads were removed.**
|
|
238
|
+
Their custom reward hooks and independent filtering rules created alternate paths around the canonical rollout checks.
|
|
239
|
+
Callers now mint once and every downstream transform reads the same reward and split fields.
|
|
240
|
+
- **`extractVerifiableRewardsFromRecords` gated only its judge-fallback branch.** The deterministic
|
|
241
|
+
branch — the highest-credibility channel the module emits, `determinism: 'deterministic'`,
|
|
242
|
+
`confidence: 1`, and what the module header calls "the RL training signal" — returned the layer
|
|
243
|
+
score untouched, so a gated run carrying `outcome.raw['layer.test'] = 1.0` exported at value 1 and
|
|
244
|
+
`filterDeterministicallyRewarded` kept it. That is exactly the shape of a reward-hacked coding run:
|
|
245
|
+
`realness.gated` means the success signal was faked, and a test suite reporting green on a stubbed
|
|
246
|
+
integration IS the deterministic layer being the thing that got faked. The gate applies to that
|
|
247
|
+
channel most, not least. `value` and every `components` entry are now 0 on a gated run — zeroing
|
|
248
|
+
`value` alone would let a consumer re-weighting per source reconstruct the refused reward — and the
|
|
249
|
+
new `VerifiableReward.realnessGated` distinguishes "measured a genuine failure" from "claimed a
|
|
250
|
+
success we refuse to believe", which a bare 0 cannot.
|
|
251
|
+
- **`toPrmRows(triples, lookups)` — the deprecated 2-arg form — applied no gate and now fails
|
|
252
|
+
closed.** A `PrmTrainingTriple` carries a bare `chosenReward` number, so without the minted lines
|
|
253
|
+
the exporter has no way to learn that its chosen step belongs to a run that faked its success; the
|
|
254
|
+
rows it produced trained a process-reward model to prefer the gaming move at the exact step the
|
|
255
|
+
gaming happened. The overload is removed (TypeScript callers fail to compile) and the runtime
|
|
256
|
+
throws for everyone else.
|
|
257
|
+
- **`supervisorRunRolloutLines` no longer writes the reward pair by hand.** `reward` and
|
|
258
|
+
`realness_gated` come out of one call in the module that owns the gate — `rolloutRewardFields` for
|
|
259
|
+
`mintRolloutRows`, `unscreenedRewardFields` for a producer with a score but no `RunRecord` behind
|
|
260
|
+
it. Two minting doors is the same class of defect as two reward derivations; there is now one
|
|
261
|
+
writer of the pair, so a future third door cannot state one field and forget the other.
|
|
262
|
+
- **`EvalTraceStore.compareRuns` counted a gamed run as a silent zero.** Gating `runScore` (second
|
|
263
|
+
pass, above) fixed few-shot seeding and quietly changed this: a gamed run entered the paired
|
|
264
|
+
comparison at 0, which reads as "this candidate failed the scenario" when what happened is "this
|
|
265
|
+
candidate's result is not evidence". Gated runs are now excluded and counted in
|
|
266
|
+
`CandidateComparison.realnessGatedRuns` — the same never-a-silent-0 rule `passRate` follows.
|
|
267
|
+
|
|
268
|
+
#### Deliberately NOT gated
|
|
269
|
+
|
|
270
|
+
`rl/reward-hacking.ts` reads the deterministic reward through the new
|
|
271
|
+
`VerifiableRewardExtractionOptions.applyRealnessGate: false`, which preserves its previous behaviour
|
|
272
|
+
exactly. Its `judge_drift` and `reward_disagreement` signals measure the GAP between the judge reward
|
|
273
|
+
and the deterministic one; a deterministic reward another gate already forced to 0 opens that gap by
|
|
274
|
+
construction on the gamed population, so the detector would fire on its own input rather than on
|
|
275
|
+
evidence it found. Same reasoning as its ungated `DEFAULT_PROXY`. The option defaults to `true` and an
|
|
276
|
+
empty options object gates — the opt-out is explicit and greppable.
|
|
277
|
+
|
|
278
|
+
### Known, not fixed here
|
|
279
|
+
|
|
280
|
+
- `description-length-gate.ts` gives a gated run claiming `score: 1.0` the largest possible improvement
|
|
281
|
+
to its objective, and `product-benchmark/export.ts` publishes a gated run with `pass: true`. Each
|
|
282
|
+
site carries a comment naming the hole.
|
|
283
|
+
- `extractVerifiableReward(report)` — the `VerificationReport` signature — cannot gate and does not
|
|
284
|
+
claim to: `realness` lives on the `RunRecord`, not on the report. Documented on the function.
|
|
285
|
+
|
|
7
286
|
## [0.128.2] - 2026-07-25 - current core contract
|
|
8
287
|
|
|
9
288
|
### Changed
|
package/README.md
CHANGED
|
@@ -24,6 +24,24 @@ Model calls occur only through the clients and agents you configure.
|
|
|
24
24
|
pnpm add @tangle-network/agent-eval
|
|
25
25
|
```
|
|
26
26
|
|
|
27
|
+
## Configure Model Calls
|
|
28
|
+
|
|
29
|
+
Benchmarks, user drivers, executors, built-in judges, completion checkers, and judge adapters accept the same `ChatClient`.
|
|
30
|
+
|
|
31
|
+
```ts
|
|
32
|
+
import { createChatClient } from '@tangle-network/agent-eval'
|
|
33
|
+
|
|
34
|
+
const chat = createChatClient({
|
|
35
|
+
transport: 'router',
|
|
36
|
+
apiKey: process.env.TANGLE_API_KEY!,
|
|
37
|
+
defaultModel: 'openai/gpt-4.1',
|
|
38
|
+
maximumAttempts: 3,
|
|
39
|
+
})
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Use `direct-provider` for an OpenAI-compatible endpoint, `cli-bridge` for a local subscription, `sandbox-sdk` for Sandbox, or `custom` to adapt another SDK.
|
|
43
|
+
A custom adapter must return `ChatResponse` and declare `maximumAttempts` before a capped cost ledger can dispatch it.
|
|
44
|
+
|
|
27
45
|
The official optimizers use the Python bridge.
|
|
28
46
|
Install only the optimizer you plan to run:
|
|
29
47
|
|
|
@@ -346,6 +364,7 @@ See [concepts](./docs/concepts.md), [customer paths](./docs/customer-journeys.md
|
|
|
346
364
|
|---|---|
|
|
347
365
|
| `@tangle-network/agent-eval/contract` | Define an evaluation, run it, improve with a custom candidate generator, and analyze runs. |
|
|
348
366
|
| `@tangle-network/agent-eval/campaign` | Control campaigns, official optimization methods, comparisons, storage, and release rules. |
|
|
367
|
+
| `@tangle-network/agent-eval/ledger-core` | Generic hash-chained append-only journal: idempotent append, chain verification, replay-to-projection, cross-process locking. |
|
|
349
368
|
| `@tangle-network/agent-eval/reporting` | Statistical comparisons and report rendering. |
|
|
350
369
|
| `@tangle-network/agent-eval/analyst` | Model-assisted failure analysis. |
|
|
351
370
|
| `@tangle-network/agent-eval/traces` | Store, replay, and inspect structured traces. |
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
import { n as observedScore } from "./reward-nw2xZGZG.js";
|
|
2
|
+
//#region src/rl/active-curriculum.ts
|
|
3
|
+
/**
|
|
4
|
+
* Adaptive curriculum / active scenario selection.
|
|
5
|
+
*
|
|
6
|
+
* Fixed scenario sets waste sample budget on cells the policy already
|
|
7
|
+
* passes (no information left) and cells the policy never passes (no
|
|
8
|
+
* gradient available either). Active learning over scenarios fixes this
|
|
9
|
+
* by allocating the next sample budget to cells where the policy's
|
|
10
|
+
* outcome is *uncertain* — those carry the most decision-relevant signal.
|
|
11
|
+
*
|
|
12
|
+
* This module ships two complementary strategies:
|
|
13
|
+
*
|
|
14
|
+
* 1. **Variance-based** — score each (variant, scenario) cell by the
|
|
15
|
+
* empirical variance of past observations. Allocate next-round budget
|
|
16
|
+
* proportional to variance. Standard active-learning-by-uncertainty
|
|
17
|
+
* heuristic; works well when the policy is non-deterministic and
|
|
18
|
+
* cells differ in observation noise.
|
|
19
|
+
*
|
|
20
|
+
* 2. **Bandit-based (Thompson sampling)** — model each (variant,
|
|
21
|
+
* scenario) cell as a Beta-Bernoulli arm; sample a posterior; pick
|
|
22
|
+
* cells whose posterior mean is closest to the per-scenario decision
|
|
23
|
+
* threshold. The right primitive when scenarios are
|
|
24
|
+
* "pass/fail" rather than continuous, and when promotion gates fire
|
|
25
|
+
* at a known threshold (e.g., 0.5).
|
|
26
|
+
*
|
|
27
|
+
* The output is a *next-round budget allocation* — a list of (variant,
|
|
28
|
+
* scenario, count) triples. The consumer's matrix runner consumes the
|
|
29
|
+
* allocation, runs those cells, feeds the new observations back. Loop.
|
|
30
|
+
*
|
|
31
|
+
* Out of scope (deliberate): scenario *generation* — that's the
|
|
32
|
+
* adversarial primitive's job. This module allocates over an existing
|
|
33
|
+
* scenario pool.
|
|
34
|
+
*/
|
|
35
|
+
/**
|
|
36
|
+
* Variance-proportional allocation. For each cell, estimate variance from
|
|
37
|
+
* past observations + a prior, then allocate the budget proportional to
|
|
38
|
+
* (sqrt(variance) + 1/sqrt(n)) — a classical optimal-allocation rule
|
|
39
|
+
* (Neyman 1934) that balances "explore noisy cells" with "explore
|
|
40
|
+
* under-sampled cells."
|
|
41
|
+
*/
|
|
42
|
+
function varianceBasedCurriculum(observations, candidateCells, opts) {
|
|
43
|
+
const variancePrior = opts.variancePrior ?? .05;
|
|
44
|
+
const floor = opts.floorPerCell ?? 1;
|
|
45
|
+
const budget = opts.budget;
|
|
46
|
+
const grouped = /* @__PURE__ */ new Map();
|
|
47
|
+
for (const o of observations) {
|
|
48
|
+
const k = `${o.variantId}::${o.scenarioId}`;
|
|
49
|
+
const arr = grouped.get(k) ?? [];
|
|
50
|
+
arr.push(o.score);
|
|
51
|
+
grouped.set(k, arr);
|
|
52
|
+
}
|
|
53
|
+
const cellStats = candidateCells.map((c) => {
|
|
54
|
+
const k = `${c.variantId}::${c.scenarioId}`;
|
|
55
|
+
const samples = grouped.get(k) ?? [];
|
|
56
|
+
const n = samples.length;
|
|
57
|
+
const mean = n === 0 ? .5 : samples.reduce((s, v) => s + v, 0) / n;
|
|
58
|
+
const variance = n < 2 ? variancePrior : samples.reduce((s, v) => s + (v - mean) ** 2, 0) / (n - 1) + variancePrior;
|
|
59
|
+
const weight = Math.sqrt(variance) + 1 / Math.sqrt(Math.max(1, n));
|
|
60
|
+
return {
|
|
61
|
+
variantId: c.variantId,
|
|
62
|
+
scenarioId: c.scenarioId,
|
|
63
|
+
n,
|
|
64
|
+
mean,
|
|
65
|
+
variance,
|
|
66
|
+
weight
|
|
67
|
+
};
|
|
68
|
+
});
|
|
69
|
+
const floorTotal = floor * cellStats.length;
|
|
70
|
+
if (floorTotal >= budget) {
|
|
71
|
+
const each = Math.max(1, Math.floor(budget / Math.max(1, cellStats.length)));
|
|
72
|
+
return cellStats.map((c) => ({
|
|
73
|
+
variantId: c.variantId,
|
|
74
|
+
scenarioId: c.scenarioId,
|
|
75
|
+
count: each,
|
|
76
|
+
reason: `floor allocation (budget tight; n=${c.n})`
|
|
77
|
+
}));
|
|
78
|
+
}
|
|
79
|
+
const remaining = budget - floorTotal;
|
|
80
|
+
const totalWeight = cellStats.reduce((s, c) => s + c.weight, 0);
|
|
81
|
+
return cellStats.map((c) => {
|
|
82
|
+
const proportional = totalWeight === 0 ? 0 : Math.round(c.weight / totalWeight * remaining);
|
|
83
|
+
return {
|
|
84
|
+
variantId: c.variantId,
|
|
85
|
+
scenarioId: c.scenarioId,
|
|
86
|
+
count: floor + proportional,
|
|
87
|
+
reason: `variance ${c.variance.toFixed(3)} (n=${c.n}, mean=${c.mean.toFixed(3)})`
|
|
88
|
+
};
|
|
89
|
+
});
|
|
90
|
+
}
|
|
91
|
+
/**
|
|
92
|
+
* Thompson-sampling-style allocation for pass/fail cells. For each cell:
|
|
93
|
+
*
|
|
94
|
+
* - Maintain Beta(α + passes, β + failures) posterior on pass-rate
|
|
95
|
+
* - Allocation weight ∝ exp(-((sampledMean - threshold) / σ)^2):
|
|
96
|
+
* cells whose sampled posterior straddles the decision boundary get
|
|
97
|
+
* the most weight; cells already clearly above or below get less.
|
|
98
|
+
*
|
|
99
|
+
* This is the right primitive when promotion gates fire at a known
|
|
100
|
+
* threshold and you want to sharpen the posterior near the boundary.
|
|
101
|
+
*/
|
|
102
|
+
function thompsonCurriculum(observations, candidateCells, opts) {
|
|
103
|
+
const threshold = opts.decisionThreshold ?? .5;
|
|
104
|
+
const alpha0 = opts.priorAlpha ?? 1;
|
|
105
|
+
const beta0 = opts.priorBeta ?? 1;
|
|
106
|
+
const rng = makeRng(opts.seed);
|
|
107
|
+
const grouped = /* @__PURE__ */ new Map();
|
|
108
|
+
for (const o of observations) {
|
|
109
|
+
const k = `${o.variantId}::${o.scenarioId}`;
|
|
110
|
+
const cur = grouped.get(k) ?? {
|
|
111
|
+
passes: 0,
|
|
112
|
+
failures: 0
|
|
113
|
+
};
|
|
114
|
+
if (o.pass ?? o.score >= threshold) cur.passes += 1;
|
|
115
|
+
else cur.failures += 1;
|
|
116
|
+
grouped.set(k, cur);
|
|
117
|
+
}
|
|
118
|
+
const stats = candidateCells.map((c) => {
|
|
119
|
+
const k = `${c.variantId}::${c.scenarioId}`;
|
|
120
|
+
const cur = grouped.get(k) ?? {
|
|
121
|
+
passes: 0,
|
|
122
|
+
failures: 0
|
|
123
|
+
};
|
|
124
|
+
const a = alpha0 + cur.passes;
|
|
125
|
+
const b = beta0 + cur.failures;
|
|
126
|
+
const sampled = sampleBeta(a, b, rng);
|
|
127
|
+
const distance = Math.abs(sampled - threshold);
|
|
128
|
+
const variance = a * b / ((a + b) ** 2 * (a + b + 1));
|
|
129
|
+
const sigma = Math.max(.05, Math.sqrt(variance));
|
|
130
|
+
const weight = Math.exp(-((distance / sigma) ** 2));
|
|
131
|
+
return {
|
|
132
|
+
variantId: c.variantId,
|
|
133
|
+
scenarioId: c.scenarioId,
|
|
134
|
+
n: cur.passes + cur.failures,
|
|
135
|
+
sampled,
|
|
136
|
+
sigma,
|
|
137
|
+
weight,
|
|
138
|
+
a,
|
|
139
|
+
b
|
|
140
|
+
};
|
|
141
|
+
});
|
|
142
|
+
const totalWeight = stats.reduce((s, c) => s + c.weight, 0);
|
|
143
|
+
return stats.map((c) => {
|
|
144
|
+
const proportional = totalWeight === 0 ? 0 : Math.round(c.weight / totalWeight * opts.budget);
|
|
145
|
+
return {
|
|
146
|
+
variantId: c.variantId,
|
|
147
|
+
scenarioId: c.scenarioId,
|
|
148
|
+
count: Math.max(0, proportional),
|
|
149
|
+
reason: `Beta(${c.a.toFixed(1)},${c.b.toFixed(1)}) sample=${c.sampled.toFixed(3)} (target ${threshold})`
|
|
150
|
+
};
|
|
151
|
+
});
|
|
152
|
+
}
|
|
153
|
+
/** Convenience: extract `CellObservation[]` directly from `RunRecord[]`. */
|
|
154
|
+
function observationsFromRunRecords(runs, opts = {}) {
|
|
155
|
+
const threshold = opts.passThreshold ?? .5;
|
|
156
|
+
const useHoldout = opts.useHoldout ?? true;
|
|
157
|
+
const out = [];
|
|
158
|
+
for (const r of runs) {
|
|
159
|
+
if (!r.scenarioId) continue;
|
|
160
|
+
const score = observedScore(r, useHoldout ? "holdout" : "search");
|
|
161
|
+
if (typeof score !== "number" || !Number.isFinite(score)) continue;
|
|
162
|
+
out.push({
|
|
163
|
+
variantId: r.candidateId,
|
|
164
|
+
scenarioId: r.scenarioId,
|
|
165
|
+
score,
|
|
166
|
+
pass: score >= threshold
|
|
167
|
+
});
|
|
168
|
+
}
|
|
169
|
+
return out;
|
|
170
|
+
}
|
|
171
|
+
function makeRng(seed) {
|
|
172
|
+
if (seed === void 0) return Math.random;
|
|
173
|
+
let s = seed >>> 0;
|
|
174
|
+
return () => {
|
|
175
|
+
s = s + 1831565813 >>> 0;
|
|
176
|
+
let t = s;
|
|
177
|
+
t = Math.imul(t ^ t >>> 15, t | 1);
|
|
178
|
+
t ^= t + Math.imul(t ^ t >>> 7, t | 61);
|
|
179
|
+
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
180
|
+
};
|
|
181
|
+
}
|
|
182
|
+
/**
|
|
183
|
+
* Sample from Beta(α, β) via the Marsaglia–Tsang method using two Gamma
|
|
184
|
+
* variates. Accuracy is good for α, β > 1; we floor the parameters at 1
|
|
185
|
+
* to avoid degenerate cases.
|
|
186
|
+
*/
|
|
187
|
+
function sampleBeta(alpha, beta, rng) {
|
|
188
|
+
const a = Math.max(1, alpha);
|
|
189
|
+
const b = Math.max(1, beta);
|
|
190
|
+
const x = sampleGamma(a, rng);
|
|
191
|
+
return x / (x + sampleGamma(b, rng));
|
|
192
|
+
}
|
|
193
|
+
function sampleGamma(shape, rng) {
|
|
194
|
+
const d = shape - 1 / 3;
|
|
195
|
+
const c = 1 / Math.sqrt(9 * d);
|
|
196
|
+
while (true) {
|
|
197
|
+
let x;
|
|
198
|
+
let v;
|
|
199
|
+
do {
|
|
200
|
+
const u1 = rng() || 1e-12;
|
|
201
|
+
const u2 = rng() || 1e-12;
|
|
202
|
+
x = Math.sqrt(-2 * Math.log(u1)) * Math.cos(2 * Math.PI * u2);
|
|
203
|
+
v = 1 + c * x;
|
|
204
|
+
} while (v <= 0);
|
|
205
|
+
v = v * v * v;
|
|
206
|
+
const u = rng();
|
|
207
|
+
if (u < 1 - .0331 * x ** 4) return d * v;
|
|
208
|
+
if (Math.log(u) < .5 * x * x + d * (1 - v + Math.log(v))) return d * v;
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
//#endregion
|
|
212
|
+
export { thompsonCurriculum as n, varianceBasedCurriculum as r, observationsFromRunRecords as t };
|
|
213
|
+
|
|
214
|
+
//# sourceMappingURL=active-curriculum-C4mk67HP.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"active-curriculum-C4mk67HP.js","names":[],"sources":["../src/rl/active-curriculum.ts"],"sourcesContent":["/**\n * Adaptive curriculum / active scenario selection.\n *\n * Fixed scenario sets waste sample budget on cells the policy already\n * passes (no information left) and cells the policy never passes (no\n * gradient available either). Active learning over scenarios fixes this\n * by allocating the next sample budget to cells where the policy's\n * outcome is *uncertain* — those carry the most decision-relevant signal.\n *\n * This module ships two complementary strategies:\n *\n * 1. **Variance-based** — score each (variant, scenario) cell by the\n * empirical variance of past observations. Allocate next-round budget\n * proportional to variance. Standard active-learning-by-uncertainty\n * heuristic; works well when the policy is non-deterministic and\n * cells differ in observation noise.\n *\n * 2. **Bandit-based (Thompson sampling)** — model each (variant,\n * scenario) cell as a Beta-Bernoulli arm; sample a posterior; pick\n * cells whose posterior mean is closest to the per-scenario decision\n * threshold. The right primitive when scenarios are\n * \"pass/fail\" rather than continuous, and when promotion gates fire\n * at a known threshold (e.g., 0.5).\n *\n * The output is a *next-round budget allocation* — a list of (variant,\n * scenario, count) triples. The consumer's matrix runner consumes the\n * allocation, runs those cells, feeds the new observations back. Loop.\n *\n * Out of scope (deliberate): scenario *generation* — that's the\n * adversarial primitive's job. This module allocates over an existing\n * scenario pool.\n */\n\nimport { observedScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\n\nexport interface CellObservation {\n variantId: string\n scenarioId: string\n /** Observed score in [0, 1]. */\n score: number\n /** For Bernoulli arms — derive from the score with a threshold if needed. */\n pass?: boolean\n}\n\nexport interface CurriculumAllocation {\n variantId: string\n scenarioId: string\n /** How many additional reps to run on this cell. */\n count: number\n /** Strategy-specific reason for the allocation. */\n reason: string\n}\n\nexport interface VarianceCurriculumOptions {\n /** Total reps to allocate across all cells. */\n budget: number\n /**\n * Smoothing prior on variance — keeps the allocator from concentrating\n * on a cell with one observation just because its 1-sample variance is\n * 0. Default 0.05.\n */\n variancePrior?: number\n /**\n * Minimum reps per cell — even when the variance estimate is low, give\n * every cell at least this many. Default 1.\n */\n floorPerCell?: number\n}\n\n/**\n * Variance-proportional allocation. For each cell, estimate variance from\n * past observations + a prior, then allocate the budget proportional to\n * (sqrt(variance) + 1/sqrt(n)) — a classical optimal-allocation rule\n * (Neyman 1934) that balances \"explore noisy cells\" with \"explore\n * under-sampled cells.\"\n */\nexport function varianceBasedCurriculum(\n observations: CellObservation[],\n candidateCells: Array<{ variantId: string; scenarioId: string }>,\n opts: VarianceCurriculumOptions,\n): CurriculumAllocation[] {\n const variancePrior = opts.variancePrior ?? 0.05\n const floor = opts.floorPerCell ?? 1\n const budget = opts.budget\n\n const grouped = new Map<string, number[]>()\n for (const o of observations) {\n const k = `${o.variantId}::${o.scenarioId}`\n const arr = grouped.get(k) ?? []\n arr.push(o.score)\n grouped.set(k, arr)\n }\n\n const cellStats = candidateCells.map((c) => {\n const k = `${c.variantId}::${c.scenarioId}`\n const samples = grouped.get(k) ?? []\n const n = samples.length\n const mean = n === 0 ? 0.5 : samples.reduce((s, v) => s + v, 0) / n\n const variance =\n n < 2\n ? variancePrior\n : samples.reduce((s, v) => s + (v - mean) ** 2, 0) / (n - 1) + variancePrior\n // Neyman optimal allocation: weight ∝ √variance; add √(1/n) to break\n // ties toward under-sampled cells.\n const weight = Math.sqrt(variance) + 1 / Math.sqrt(Math.max(1, n))\n return { variantId: c.variantId, scenarioId: c.scenarioId, n, mean, variance, weight }\n })\n\n // Reserve floor*N for the floor; allocate the rest proportional to weight.\n const floorTotal = floor * cellStats.length\n if (floorTotal >= budget) {\n const each = Math.max(1, Math.floor(budget / Math.max(1, cellStats.length)))\n return cellStats.map((c) => ({\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: each,\n reason: `floor allocation (budget tight; n=${c.n})`,\n }))\n }\n const remaining = budget - floorTotal\n const totalWeight = cellStats.reduce((s, c) => s + c.weight, 0)\n return cellStats.map((c) => {\n const proportional = totalWeight === 0 ? 0 : Math.round((c.weight / totalWeight) * remaining)\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: floor + proportional,\n reason: `variance ${c.variance.toFixed(3)} (n=${c.n}, mean=${c.mean.toFixed(3)})`,\n }\n })\n}\n\nexport interface ThompsonCurriculumOptions {\n budget: number\n /**\n * The per-scenario decision threshold. Cells whose posterior mean is\n * closest to this get the most budget — that's where the next observation\n * has the highest information value for the gate decision. Default 0.5.\n */\n decisionThreshold?: number\n /** Beta prior parameters. Default α=β=1 (uniform). */\n priorAlpha?: number\n priorBeta?: number\n /** Seed the Thompson sampler. Default unset (Math.random). */\n seed?: number\n}\n\n/**\n * Thompson-sampling-style allocation for pass/fail cells. For each cell:\n *\n * - Maintain Beta(α + passes, β + failures) posterior on pass-rate\n * - Allocation weight ∝ exp(-((sampledMean - threshold) / σ)^2):\n * cells whose sampled posterior straddles the decision boundary get\n * the most weight; cells already clearly above or below get less.\n *\n * This is the right primitive when promotion gates fire at a known\n * threshold and you want to sharpen the posterior near the boundary.\n */\nexport function thompsonCurriculum(\n observations: CellObservation[],\n candidateCells: Array<{ variantId: string; scenarioId: string }>,\n opts: ThompsonCurriculumOptions,\n): CurriculumAllocation[] {\n const threshold = opts.decisionThreshold ?? 0.5\n const alpha0 = opts.priorAlpha ?? 1\n const beta0 = opts.priorBeta ?? 1\n const rng = makeRng(opts.seed)\n\n const grouped = new Map<string, { passes: number; failures: number }>()\n for (const o of observations) {\n const k = `${o.variantId}::${o.scenarioId}`\n const cur = grouped.get(k) ?? { passes: 0, failures: 0 }\n const pass = o.pass ?? o.score >= threshold\n if (pass) cur.passes += 1\n else cur.failures += 1\n grouped.set(k, cur)\n }\n\n const stats = candidateCells.map((c) => {\n const k = `${c.variantId}::${c.scenarioId}`\n const cur = grouped.get(k) ?? { passes: 0, failures: 0 }\n const a = alpha0 + cur.passes\n const b = beta0 + cur.failures\n // Sample a single Beta draw — the Thompson signal.\n const sampled = sampleBeta(a, b, rng)\n const distance = Math.abs(sampled - threshold)\n // Information-near-threshold weight: closer = higher.\n // Use Gaussian-shaped kernel with σ tuned to posterior std.\n const variance = (a * b) / ((a + b) ** 2 * (a + b + 1))\n const sigma = Math.max(0.05, Math.sqrt(variance))\n const weight = Math.exp(-((distance / sigma) ** 2))\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n n: cur.passes + cur.failures,\n sampled,\n sigma,\n weight,\n a,\n b,\n }\n })\n\n const totalWeight = stats.reduce((s, c) => s + c.weight, 0)\n return stats.map((c) => {\n const proportional = totalWeight === 0 ? 0 : Math.round((c.weight / totalWeight) * opts.budget)\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: Math.max(0, proportional),\n reason: `Beta(${c.a.toFixed(1)},${c.b.toFixed(1)}) sample=${c.sampled.toFixed(3)} (target ${threshold})`,\n }\n })\n}\n\n/** Convenience: extract `CellObservation[]` directly from `RunRecord[]`. */\nexport function observationsFromRunRecords(\n runs: RunRecord[],\n opts: { passThreshold?: number; useHoldout?: boolean } = {},\n): CellObservation[] {\n const threshold = opts.passThreshold ?? 0.5\n const useHoldout = opts.useHoldout ?? true\n const out: CellObservation[] = []\n for (const r of runs) {\n if (!r.scenarioId) continue\n // Ungated on purpose, and the precedence is caller policy, not a default:\n // `useHoldout: false` means \"score this curriculum on the search split when\n // both exist\". This feeds sampling COUNTS, not an exported reward. Known\n // risk: a gamed run's high score inflates the cell's Beta posterior, so the\n // curriculum stops sampling a cell it wrongly believes is solved. The fix\n // for that is an upstream filter on gated records — zeroing the score here\n // would push the posterior the opposite way and be equally wrong.\n const score = observedScore(r, useHoldout ? 'holdout' : 'search')\n if (typeof score !== 'number' || !Number.isFinite(score)) continue\n out.push({\n variantId: r.candidateId,\n scenarioId: r.scenarioId,\n score,\n pass: score >= threshold,\n })\n }\n return out\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction makeRng(seed?: number): () => number {\n if (seed === undefined) return Math.random\n let s = seed >>> 0\n return () => {\n s = (s + 0x6d2b79f5) >>> 0\n let t = s\n t = Math.imul(t ^ (t >>> 15), t | 1)\n t ^= t + Math.imul(t ^ (t >>> 7), t | 61)\n return ((t ^ (t >>> 14)) >>> 0) / 4294967296\n }\n}\n\n/**\n * Sample from Beta(α, β) via the Marsaglia–Tsang method using two Gamma\n * variates. Accuracy is good for α, β > 1; we floor the parameters at 1\n * to avoid degenerate cases.\n */\nfunction sampleBeta(alpha: number, beta: number, rng: () => number): number {\n const a = Math.max(1, alpha)\n const b = Math.max(1, beta)\n const x = sampleGamma(a, rng)\n const y = sampleGamma(b, rng)\n return x / (x + y)\n}\n\nfunction sampleGamma(shape: number, rng: () => number): number {\n // Marsaglia–Tsang for shape ≥ 1.\n const d = shape - 1 / 3\n const c = 1 / Math.sqrt(9 * d)\n while (true) {\n let x: number\n let v: number\n do {\n const u1 = rng() || 1e-12\n const u2 = rng() || 1e-12\n // Box-Muller for a normal sample.\n x = Math.sqrt(-2 * Math.log(u1)) * Math.cos(2 * Math.PI * u2)\n v = 1 + c * x\n } while (v <= 0)\n v = v * v * v\n const u = rng()\n if (u < 1 - 0.0331 * x ** 4) return d * v\n if (Math.log(u) < 0.5 * x * x + d * (1 - v + Math.log(v))) return d * v\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA6EA,SAAgB,wBACd,cACA,gBACA,MACwB;CACxB,MAAM,gBAAgB,KAAK,iBAAiB;CAC5C,MAAM,QAAQ,KAAK,gBAAgB;CACnC,MAAM,SAAS,KAAK;CAEpB,MAAM,0BAAU,IAAI,IAAsB;CAC1C,KAAK,MAAM,KAAK,cAAc;EAC5B,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,MAAM,QAAQ,IAAI,CAAC,KAAK,CAAC;EAC/B,IAAI,KAAK,EAAE,KAAK;EAChB,QAAQ,IAAI,GAAG,GAAG;CACpB;CAEA,MAAM,YAAY,eAAe,KAAK,MAAM;EAC1C,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,UAAU,QAAQ,IAAI,CAAC,KAAK,CAAC;EACnC,MAAM,IAAI,QAAQ;EAClB,MAAM,OAAO,MAAM,IAAI,KAAM,QAAQ,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;EAClE,MAAM,WACJ,IAAI,IACA,gBACA,QAAQ,QAAQ,GAAG,MAAM,KAAK,IAAI,SAAS,GAAG,CAAC,KAAK,IAAI,KAAK;EAGnE,MAAM,SAAS,KAAK,KAAK,QAAQ,IAAI,IAAI,KAAK,KAAK,KAAK,IAAI,GAAG,CAAC,CAAC;EACjE,OAAO;GAAE,WAAW,EAAE;GAAW,YAAY,EAAE;GAAY;GAAG;GAAM;GAAU;EAAO;CACvF,CAAC;CAGD,MAAM,aAAa,QAAQ,UAAU;CACrC,IAAI,cAAc,QAAQ;EACxB,MAAM,OAAO,KAAK,IAAI,GAAG,KAAK,MAAM,SAAS,KAAK,IAAI,GAAG,UAAU,MAAM,CAAC,CAAC;EAC3E,OAAO,UAAU,KAAK,OAAO;GAC3B,WAAW,EAAE;GACb,YAAY,EAAE;GACd,OAAO;GACP,QAAQ,qCAAqC,EAAE,EAAE;EACnD,EAAE;CACJ;CACA,MAAM,YAAY,SAAS;CAC3B,MAAM,cAAc,UAAU,QAAQ,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;CAC9D,OAAO,UAAU,KAAK,MAAM;EAC1B,MAAM,eAAe,gBAAgB,IAAI,IAAI,KAAK,MAAO,EAAE,SAAS,cAAe,SAAS;EAC5F,OAAO;GACL,WAAW,EAAE;GACb,YAAY,EAAE;GACd,OAAO,QAAQ;GACf,QAAQ,YAAY,EAAE,SAAS,QAAQ,CAAC,EAAE,MAAM,EAAE,EAAE,SAAS,EAAE,KAAK,QAAQ,CAAC,EAAE;EACjF;CACF,CAAC;AACH;;;;;;;;;;;;AA4BA,SAAgB,mBACd,cACA,gBACA,MACwB;CACxB,MAAM,YAAY,KAAK,qBAAqB;CAC5C,MAAM,SAAS,KAAK,cAAc;CAClC,MAAM,QAAQ,KAAK,aAAa;CAChC,MAAM,MAAM,QAAQ,KAAK,IAAI;CAE7B,MAAM,0BAAU,IAAI,IAAkD;CACtE,KAAK,MAAM,KAAK,cAAc;EAC5B,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,MAAM,QAAQ,IAAI,CAAC,KAAK;GAAE,QAAQ;GAAG,UAAU;EAAE;EAEvD,IADa,EAAE,QAAQ,EAAE,SAAS,WACxB,IAAI,UAAU;OACnB,IAAI,YAAY;EACrB,QAAQ,IAAI,GAAG,GAAG;CACpB;CAEA,MAAM,QAAQ,eAAe,KAAK,MAAM;EACtC,MAAM,IAAI,GAAG,EAAE,UAAU,IAAI,EAAE;EAC/B,MAAM,MAAM,QAAQ,IAAI,CAAC,KAAK;GAAE,QAAQ;GAAG,UAAU;EAAE;EACvD,MAAM,IAAI,SAAS,IAAI;EACvB,MAAM,IAAI,QAAQ,IAAI;EAEtB,MAAM,UAAU,WAAW,GAAG,GAAG,GAAG;EACpC,MAAM,WAAW,KAAK,IAAI,UAAU,SAAS;EAG7C,MAAM,WAAY,IAAI,MAAO,IAAI,MAAM,KAAK,IAAI,IAAI;EACpD,MAAM,QAAQ,KAAK,IAAI,KAAM,KAAK,KAAK,QAAQ,CAAC;EAChD,MAAM,SAAS,KAAK,IAAI,GAAG,WAAW,UAAU,EAAE;EAClD,OAAO;GACL,WAAW,EAAE;GACb,YAAY,EAAE;GACd,GAAG,IAAI,SAAS,IAAI;GACpB;GACA;GACA;GACA;GACA;EACF;CACF,CAAC;CAED,MAAM,cAAc,MAAM,QAAQ,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;CAC1D,OAAO,MAAM,KAAK,MAAM;EACtB,MAAM,eAAe,gBAAgB,IAAI,IAAI,KAAK,MAAO,EAAE,SAAS,cAAe,KAAK,MAAM;EAC9F,OAAO;GACL,WAAW,EAAE;GACb,YAAY,EAAE;GACd,OAAO,KAAK,IAAI,GAAG,YAAY;GAC/B,QAAQ,QAAQ,EAAE,EAAE,QAAQ,CAAC,EAAE,GAAG,EAAE,EAAE,QAAQ,CAAC,EAAE,WAAW,EAAE,QAAQ,QAAQ,CAAC,EAAE,WAAW,UAAU;EACxG;CACF,CAAC;AACH;;AAGA,SAAgB,2BACd,MACA,OAAyD,CAAC,GACvC;CACnB,MAAM,YAAY,KAAK,iBAAiB;CACxC,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,MAAyB,CAAC;CAChC,KAAK,MAAM,KAAK,MAAM;EACpB,IAAI,CAAC,EAAE,YAAY;EAQnB,MAAM,QAAQ,cAAc,GAAG,aAAa,YAAY,QAAQ;EAChE,IAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;EAC1D,IAAI,KAAK;GACP,WAAW,EAAE;GACb,YAAY,EAAE;GACd;GACA,MAAM,SAAS;EACjB,CAAC;CACH;CACA,OAAO;AACT;AAIA,SAAS,QAAQ,MAA6B;CAC5C,IAAI,SAAS,KAAA,GAAW,OAAO,KAAK;CACpC,IAAI,IAAI,SAAS;CACjB,aAAa;EACX,IAAK,IAAI,eAAgB;EACzB,IAAI,IAAI;EACR,IAAI,KAAK,KAAK,IAAK,MAAM,IAAK,IAAI,CAAC;EACnC,KAAK,IAAI,KAAK,KAAK,IAAK,MAAM,GAAI,IAAI,EAAE;EACxC,SAAS,IAAK,MAAM,QAAS,KAAK;CACpC;AACF;;;;;;AAOA,SAAS,WAAW,OAAe,MAAc,KAA2B;CAC1E,MAAM,IAAI,KAAK,IAAI,GAAG,KAAK;CAC3B,MAAM,IAAI,KAAK,IAAI,GAAG,IAAI;CAC1B,MAAM,IAAI,YAAY,GAAG,GAAG;CAE5B,OAAO,KAAK,IADF,YAAY,GAAG,GACT;AAClB;AAEA,SAAS,YAAY,OAAe,KAA2B;CAE7D,MAAM,IAAI,QAAQ,IAAI;CACtB,MAAM,IAAI,IAAI,KAAK,KAAK,IAAI,CAAC;CAC7B,OAAO,MAAM;EACX,IAAI;EACJ,IAAI;EACJ,GAAG;GACD,MAAM,KAAK,IAAI,KAAK;GACpB,MAAM,KAAK,IAAI,KAAK;GAEpB,IAAI,KAAK,KAAK,KAAK,KAAK,IAAI,EAAE,CAAC,IAAI,KAAK,IAAI,IAAI,KAAK,KAAK,EAAE;GAC5D,IAAI,IAAI,IAAI;EACd,SAAS,KAAK;EACd,IAAI,IAAI,IAAI;EACZ,MAAM,IAAI,IAAI;EACd,IAAI,IAAI,IAAI,QAAS,KAAK,GAAG,OAAO,IAAI;EACxC,IAAI,KAAK,IAAI,CAAC,IAAI,KAAM,IAAI,IAAI,KAAK,IAAI,IAAI,KAAK,IAAI,CAAC,IAAI,OAAO,IAAI;CACxE;AACF"}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
//#region src/rl/adversarial.d.ts
|
|
2
|
+
/**
|
|
3
|
+
* Adversarial mutation contract.
|
|
4
|
+
*
|
|
5
|
+
* `AdversarialMutation<S>` is the scenario-mutation strategy the fuzz harness
|
|
6
|
+
* (`fuzzAgent`, src/fuzz) drives: paraphrase, edge-case substitution, or
|
|
7
|
+
* compositional combination of a scenario the policy currently passes, looking
|
|
8
|
+
* for the tail inputs that break it. The harness supplies the loop; consumers
|
|
9
|
+
* supply the mutations and the failure detector.
|
|
10
|
+
*/
|
|
11
|
+
interface AdversarialMutation<S> {
|
|
12
|
+
id: string;
|
|
13
|
+
/**
|
|
14
|
+
* Mutate one scenario. Return null to skip; return one or more new
|
|
15
|
+
* scenarios. The harness deduplicates by `mutateScenarioId(scenario)`.
|
|
16
|
+
*/
|
|
17
|
+
mutate(parent: S, rng: () => number): Promise<S[]> | S[];
|
|
18
|
+
}
|
|
19
|
+
//#endregion
|
|
20
|
+
export { AdversarialMutation as t };
|
|
21
|
+
//# sourceMappingURL=adversarial-smnADNFS.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"adversarial-smnADNFS.d.ts","names":[],"sources":["../src/rl/adversarial.ts"],"mappings":";;;;;;;;;;UAUiB,oBAAoB;EACnC;;;;;EAKA,OAAO,QAAQ,GAAG,oBAAoB,QAAQ,OAAO"}
|