@tangle-network/agent-eval 0.129.0 → 0.130.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +1 -0
- package/dist/active-curriculum-C4mk67HP.js +214 -0
- package/dist/active-curriculum-C4mk67HP.js.map +1 -0
- package/dist/adversarial-smnADNFS.d.ts +21 -0
- package/dist/adversarial-smnADNFS.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +81 -2872
- package/dist/analyst/index.d.ts.map +1 -0
- package/dist/analyst/index.js +319 -360
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyst-BkTS3C58.d.ts +89 -0
- package/dist/analyst-BkTS3C58.d.ts.map +1 -0
- package/dist/analyst-LsnNpSkm.js +152 -0
- package/dist/analyst-LsnNpSkm.js.map +1 -0
- package/dist/analyze-runs-C1CavBMk.js +1067 -0
- package/dist/analyze-runs-C1CavBMk.js.map +1 -0
- package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
- package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
- package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
- package/dist/authenticity/index.d.ts +81 -79
- package/dist/authenticity/index.d.ts.map +1 -0
- package/dist/authenticity/index.js +209 -193
- package/dist/authenticity/index.js.map +1 -1
- package/dist/baseline-HsBvw_dk.js +550 -0
- package/dist/baseline-HsBvw_dk.js.map +1 -0
- package/dist/baseline-hG3K85h4.d.ts +125 -0
- package/dist/baseline-hG3K85h4.d.ts.map +1 -0
- package/dist/belief-state/index.d.ts +448 -1188
- package/dist/belief-state/index.d.ts.map +1 -0
- package/dist/belief-state/index.js +1617 -1709
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -891
- package/dist/benchmarks/index.js +2 -60
- package/dist/benchmarks-DviOvUNr.js +754 -0
- package/dist/benchmarks-DviOvUNr.js.map +1 -0
- package/dist/builder-eval/index.d.ts +150 -662
- package/dist/builder-eval/index.d.ts.map +1 -0
- package/dist/builder-eval/index.js +356 -345
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/calibration-CNWWA6K8.js +94 -0
- package/dist/calibration-CNWWA6K8.js.map +1 -0
- package/dist/campaign/index.d.ts +5 -6381
- package/dist/campaign/index.js +3 -213
- package/dist/campaign-CBKZvQ1H.js +3885 -0
- package/dist/campaign-CBKZvQ1H.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +164 -175
- package/dist/cli.js.map +1 -1
- package/dist/client-C97NMzqi.d.ts +581 -0
- package/dist/client-C97NMzqi.d.ts.map +1 -0
- package/dist/client-CYzbdJOZ.js +637 -0
- package/dist/client-CYzbdJOZ.js.map +1 -0
- package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
- package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
- package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
- package/dist/concurrency-DIxRZF_J.js +85 -0
- package/dist/concurrency-DIxRZF_J.js.map +1 -0
- package/dist/contract/index.d.ts +645 -5565
- package/dist/contract/index.d.ts.map +1 -0
- package/dist/contract/index.js +1730 -1938
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +3 -1030
- package/dist/control.js +2 -33
- package/dist/cost-ledger-DIgQUFZZ.js +801 -0
- package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
- package/dist/dataset-BvtnC8Dc.d.ts +115 -0
- package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
- package/dist/default-registry-C-vFCSEc.js +2579 -0
- package/dist/default-registry-C-vFCSEc.js.map +1 -0
- package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
- package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
- package/dist/emitter-CPBAhxum.js +266 -0
- package/dist/emitter-CPBAhxum.js.map +1 -0
- package/dist/emitter-DGQGoLyj.d.ts +113 -0
- package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
- package/dist/errors-8YnH8WlF.js +64 -0
- package/dist/errors-8YnH8WlF.js.map +1 -0
- package/dist/errors-CEk209JS.d.ts +76 -0
- package/dist/errors-CEk209JS.d.ts.map +1 -0
- package/dist/eval-campaign-DEm6c8ru.js +349 -0
- package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
- package/dist/execution-tracks-CpgFPpS5.js +93 -0
- package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
- package/dist/exporters-q9iL-2Jf.js +148 -0
- package/dist/exporters-q9iL-2Jf.js.map +1 -0
- package/dist/extract-usage-BrQ8mCLX.js +155 -0
- package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
- package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
- package/dist/fuzz.d.ts +320 -646
- package/dist/fuzz.d.ts.map +1 -0
- package/dist/fuzz.js +670 -618
- package/dist/fuzz.js.map +1 -1
- package/dist/hf-dataset-DBJXXoY1.js +763 -0
- package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
- package/dist/hosted/index.d.ts +11 -831
- package/dist/hosted/index.d.ts.map +1 -0
- package/dist/hosted/index.js +2 -37
- package/dist/index-2JJSA6-r2.d.ts +926 -0
- package/dist/index-2JJSA6-r2.d.ts.map +1 -0
- package/dist/index-6N0aYmpW.d.ts +217 -0
- package/dist/index-6N0aYmpW.d.ts.map +1 -0
- package/dist/index-BAvgST_9.d.ts +131 -0
- package/dist/index-BAvgST_9.d.ts.map +1 -0
- package/dist/index-BvnJuTGD.d.ts +68 -0
- package/dist/index-BvnJuTGD.d.ts.map +1 -0
- package/dist/index-C61Wi7yg.d.ts +547 -0
- package/dist/index-C61Wi7yg.d.ts.map +1 -0
- package/dist/index-CAPUUKaM.d.ts +335 -0
- package/dist/index-CAPUUKaM.d.ts.map +1 -0
- package/dist/index-DE5fb3EC.d.ts +2244 -0
- package/dist/index-DE5fb3EC.d.ts.map +1 -0
- package/dist/index-DSC51roc.d.ts +102 -0
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +3755 -15555
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +11182 -11216
- package/dist/index.js.map +1 -1
- package/dist/integrity-BzRbCHzi.js +424 -0
- package/dist/integrity-BzRbCHzi.js.map +1 -0
- package/dist/integrity-rmVhXWA7.d.ts +61 -0
- package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
- package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/ledger-core/index.d.ts +2 -0
- package/dist/ledger-core/index.js +2 -0
- package/dist/ledger-core-DtZz1RG0.js +388 -0
- package/dist/ledger-core-DtZz1RG0.js.map +1 -0
- package/dist/llm-client--GR4JbZE.js +687 -0
- package/dist/llm-client--GR4JbZE.js.map +1 -0
- package/dist/llm-client-B_nIBlYo.d.ts +290 -0
- package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
- package/dist/matrix/index.d.ts +3 -155
- package/dist/matrix/index.js +2 -8
- package/dist/matrix-BzQnu2S6.js +270 -0
- package/dist/matrix-BzQnu2S6.js.map +1 -0
- package/dist/meta-eval/index.d.ts +4 -1027
- package/dist/meta-eval/index.js +393 -390
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/metrics-C9YY1OcL.js +239 -0
- package/dist/metrics-C9YY1OcL.js.map +1 -0
- package/dist/mint-yN2M2eh0.js +201 -0
- package/dist/mint-yN2M2eh0.js.map +1 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
- package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
- package/dist/multishot/index.d.ts +273 -480
- package/dist/multishot/index.d.ts.map +1 -0
- package/dist/multishot/index.js +595 -548
- package/dist/multishot/index.js.map +1 -1
- package/dist/off-policy-DvgzvtIx.js +220 -0
- package/dist/off-policy-DvgzvtIx.js.map +1 -0
- package/dist/off-policy-mskQw8Mb.d.ts +153 -0
- package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
- package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
- package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
- package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
- package/dist/outcome-store-ChBKlTd_.js +75 -0
- package/dist/outcome-store-ChBKlTd_.js.map +1 -0
- package/dist/paired-arms-D9D0wXj2.js +260 -0
- package/dist/paired-arms-D9D0wXj2.js.map +1 -0
- package/dist/pipelines/index.d.ts +95 -532
- package/dist/pipelines/index.d.ts.map +1 -0
- package/dist/pipelines/index.js +497 -478
- package/dist/pipelines/index.js.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/propose-review-control-LhIWGzHl.js +1458 -0
- package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
- package/dist/query-CJ_DX8vl.d.ts +28 -0
- package/dist/query-CJ_DX8vl.d.ts.map +1 -0
- package/dist/query-Di7eEQ79.js +83 -0
- package/dist/query-Di7eEQ79.js.map +1 -0
- package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
- package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
- package/dist/redact-7Aq1ukl-.js +107 -0
- package/dist/redact-7Aq1ukl-.js.map +1 -0
- package/dist/release-report-Cz9NKH39.js +603 -0
- package/dist/release-report-Cz9NKH39.js.map +1 -0
- package/dist/release-report-mvB2G4_J.d.ts +244 -0
- package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
- package/dist/replay-GnyotH0J.js +1741 -0
- package/dist/replay-GnyotH0J.js.map +1 -0
- package/dist/replay-OoidtG1E.d.ts +749 -0
- package/dist/replay-OoidtG1E.d.ts.map +1 -0
- package/dist/reporting.d.ts +6 -1312
- package/dist/reporting.js +6 -51
- package/dist/researcher-CwTdwXG1.d.ts +314 -0
- package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
- package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
- package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
- package/dist/reward-hacking-qipEpKvY.js +596 -0
- package/dist/reward-hacking-qipEpKvY.js.map +1 -0
- package/dist/reward-nw2xZGZG.js +137 -0
- package/dist/reward-nw2xZGZG.js.map +1 -0
- package/dist/rl.d.ts +760 -4010
- package/dist/rl.d.ts.map +1 -0
- package/dist/rl.js +2325 -1958
- package/dist/rl.js.map +1 -1
- package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
- package/dist/rollout/index.d.ts +3 -2087
- package/dist/rollout/index.js +8 -168
- package/dist/rollout-BOYjemfR.js +624 -0
- package/dist/rollout-BOYjemfR.js.map +1 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
- package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
- package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
- package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
- package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
- package/dist/run-record-BuoE80Dq.js +467 -0
- package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
- package/dist/run-record-CnZu_gjl.d.ts +357 -0
- package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
- package/dist/run-score-iEEAWiBY.js +41 -0
- package/dist/run-score-iEEAWiBY.js.map +1 -0
- package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
- package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
- package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
- package/dist/schema-BtVldJ3T.d.ts +206 -0
- package/dist/schema-BtVldJ3T.d.ts.map +1 -0
- package/dist/schema-C6DW4ZHR.js +821 -0
- package/dist/schema-C6DW4ZHR.js.map +1 -0
- package/dist/schema-CRhEY1SO.js +69 -0
- package/dist/schema-CRhEY1SO.js.map +1 -0
- package/dist/schema-Cef2cFmb.d.ts +408 -0
- package/dist/schema-Cef2cFmb.d.ts.map +1 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
- package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
- package/dist/sequential-Br0mAPHA.js +148 -0
- package/dist/sequential-Br0mAPHA.js.map +1 -0
- package/dist/sequential-CYwq6Ff_.d.ts +141 -0
- package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js +43 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -0
- package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
- package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
- package/dist/server-m5D9cvnG.js +1040 -0
- package/dist/server-m5D9cvnG.js.map +1 -0
- package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
- package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
- package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
- package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
- package/dist/statistics-Cmj6nynr.d.ts +514 -0
- package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
- package/dist/statistics-CnnxdpOg.js +1437 -0
- package/dist/statistics-CnnxdpOg.js.map +1 -0
- package/dist/store-CT9YIIve.d.ts +117 -0
- package/dist/store-CT9YIIve.d.ts.map +1 -0
- package/dist/store-CxJry_cs.d.ts +229 -0
- package/dist/store-CxJry_cs.d.ts.map +1 -0
- package/dist/storyboard/index.d.ts +84 -203
- package/dist/storyboard/index.d.ts.map +1 -0
- package/dist/storyboard/index.js +609 -542
- package/dist/storyboard/index.js.map +1 -1
- package/dist/summary-report-BNs5nmXI.js +862 -0
- package/dist/summary-report-BNs5nmXI.js.map +1 -0
- package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
- package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
- package/dist/supervisor-run/index.d.ts +2 -959
- package/dist/supervisor-run/index.js +2 -65
- package/dist/supervisor-run-_lnTLM3z.js +1679 -0
- package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
- package/dist/task-failure-attributes-CQZlB3et.js +311 -0
- package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
- package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
- package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
- package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
- package/dist/tools-BmuN627J.js +1085 -0
- package/dist/tools-BmuN627J.js.map +1 -0
- package/dist/trace-attributes.d.ts +2 -52
- package/dist/trace-attributes.js +131 -61
- package/dist/trace-attributes.js.map +1 -1
- package/dist/traces.d.ts +12 -2365
- package/dist/traces.js +12 -252
- package/dist/trajectory-D_7rLrvE.js +56 -0
- package/dist/trajectory-D_7rLrvE.js.map +1 -0
- package/dist/types-DGsxbAEd.d.ts +387 -0
- package/dist/types-DGsxbAEd.d.ts.map +1 -0
- package/dist/types-k9tZGKUg.d.ts +640 -0
- package/dist/types-k9tZGKUg.d.ts.map +1 -0
- package/dist/verdict-Dps8_okt.d.ts +37 -0
- package/dist/verdict-Dps8_okt.d.ts.map +1 -0
- package/dist/wire/index.d.ts +702 -1173
- package/dist/wire/index.d.ts.map +1 -0
- package/dist/wire/index.js +2 -81
- package/package.json +17 -9
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-2QU3YOPR.js +0 -7374
- package/dist/chunk-2QU3YOPR.js.map +0 -1
- package/dist/chunk-3OCR4R5I.js +0 -728
- package/dist/chunk-3OCR4R5I.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-56TAVBOK.js +0 -698
- package/dist/chunk-5DTSBUL2.js +0 -159
- package/dist/chunk-5DTSBUL2.js.map +0 -1
- package/dist/chunk-7FO3TNPI.js +0 -232
- package/dist/chunk-7FO3TNPI.js.map +0 -1
- package/dist/chunk-7ZZMD7UK.js +0 -386
- package/dist/chunk-7ZZMD7UK.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-BSO5JDQH.js +0 -2335
- package/dist/chunk-BSO5JDQH.js.map +0 -1
- package/dist/chunk-C6LXANRU.js +0 -1550
- package/dist/chunk-C6LXANRU.js.map +0 -1
- package/dist/chunk-DODXQREJ.js +0 -752
- package/dist/chunk-DODXQREJ.js.map +0 -1
- package/dist/chunk-DRYIUNWY.js +0 -622
- package/dist/chunk-DRYIUNWY.js.map +0 -1
- package/dist/chunk-E7QXT7SX.js +0 -183
- package/dist/chunk-E7QXT7SX.js.map +0 -1
- package/dist/chunk-EG66UGL4.js +0 -341
- package/dist/chunk-EG66UGL4.js.map +0 -1
- package/dist/chunk-FXTVJPYD.js +0 -576
- package/dist/chunk-FXTVJPYD.js.map +0 -1
- package/dist/chunk-G7MGMCZD.js +0 -153
- package/dist/chunk-G7MGMCZD.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H23X7XKK.js +0 -181
- package/dist/chunk-H23X7XKK.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HPWUNB47.js +0 -289
- package/dist/chunk-HPWUNB47.js.map +0 -1
- package/dist/chunk-IYCLP2N2.js +0 -766
- package/dist/chunk-IYCLP2N2.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JQSF5DQT.js +0 -701
- package/dist/chunk-JQSF5DQT.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-M4YBQKIJ.js +0 -1040
- package/dist/chunk-M4YBQKIJ.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-NY44NC4A.js +0 -1056
- package/dist/chunk-NY44NC4A.js.map +0 -1
- package/dist/chunk-OIUOT4QD.js +0 -44
- package/dist/chunk-OIUOT4QD.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OWN5NPMC.js +0 -152
- package/dist/chunk-OWN5NPMC.js.map +0 -1
- package/dist/chunk-P6FYH6K4.js +0 -1161
- package/dist/chunk-P6FYH6K4.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PC5DOSM7.js +0 -579
- package/dist/chunk-PC5DOSM7.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QB6BDBP2.js +0 -4464
- package/dist/chunk-QB6BDBP2.js.map +0 -1
- package/dist/chunk-RXHCETDZ.js +0 -536
- package/dist/chunk-RXHCETDZ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-SFLLL76A.js +0 -669
- package/dist/chunk-SFLLL76A.js.map +0 -1
- package/dist/chunk-SZLVEKMJ.js +0 -1446
- package/dist/chunk-SZLVEKMJ.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-T6RLYGAD.js +0 -158
- package/dist/chunk-T6RLYGAD.js.map +0 -1
- package/dist/chunk-TJVT4QFF.js +0 -911
- package/dist/chunk-TJVT4QFF.js.map +0 -1
- package/dist/chunk-TQ7LNKZ3.js +0 -136
- package/dist/chunk-TQ7LNKZ3.js.map +0 -1
- package/dist/chunk-U4L7JRPZ.js +0 -1706
- package/dist/chunk-U4L7JRPZ.js.map +0 -1
- package/dist/chunk-U4PHLT2N.js +0 -419
- package/dist/chunk-U4PHLT2N.js.map +0 -1
- package/dist/chunk-VCZ5FQYW.js +0 -928
- package/dist/chunk-VCZ5FQYW.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-WGXIEX7P.js +0 -116
- package/dist/chunk-WGXIEX7P.js.map +0 -1
- package/dist/chunk-WVATSFCP.js +0 -1553
- package/dist/chunk-WVATSFCP.js.map +0 -1
- package/dist/chunk-X4YIBDER.js +0 -1662
- package/dist/chunk-X4YIBDER.js.map +0 -1
- package/dist/chunk-YQN4ICPP.js +0 -355
- package/dist/chunk-YQN4ICPP.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZHTZ4EYI.js +0 -1212
- package/dist/chunk-ZHTZ4EYI.js.map +0 -1
- package/dist/control.js.map +0 -1
- package/dist/hosted/index.js.map +0 -1
- package/dist/matrix/index.js.map +0 -1
- package/dist/reporting.js.map +0 -1
- package/dist/rollout/index.js.map +0 -1
- package/dist/run-campaign-OJJ7CZF4.js +0 -18
- package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
- package/dist/supervisor-run/index.js.map +0 -1
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.js.map +0 -1
|
@@ -1,959 +1,2 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
* agent-eval. One JSONL line per agent invocation (a solo eval run, a
|
|
4
|
-
* supervisor episode, a worker session, a proposer shot, a judge call, an
|
|
5
|
-
* analyst pass), labeled with its task/split coordinates and a single
|
|
6
|
-
* scalar reward, carrying the FULL message transcript inline.
|
|
7
|
-
*
|
|
8
|
-
* This schema is the reconciliation of two prior producers:
|
|
9
|
-
* - agent-eval's RunRecord-joined rollout rows (PR #410): identity,
|
|
10
|
-
* provenance hashes, the realness gate travelling into the reward,
|
|
11
|
-
* trace-derived steps.
|
|
12
|
-
* - the bench rollout-ledger (agent-runtime PR #591): the wire shape —
|
|
13
|
-
* role, task.split/rep, parent_rollout_id, policy provenance, capture
|
|
14
|
-
* provenance, inline canonical chat-with-tools messages.
|
|
15
|
-
* Where the two conflicted, RunRecord-derived semantics won; the wire
|
|
16
|
-
* field names follow the ledger (snake_case). See `docs/rollout.md` for
|
|
17
|
-
* the field-by-field decision table.
|
|
18
|
-
*
|
|
19
|
-
* Messages are inlined — never referenced — because every harness store a
|
|
20
|
-
* rollout can be recovered from is mutable or garbage-collected. A line
|
|
21
|
-
* must stay a complete training/eval example on its own.
|
|
22
|
-
*
|
|
23
|
-
* `outcome.reward` is THE single scalar (null = no verdict exists — a
|
|
24
|
-
* labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart
|
|
25
|
-
* flag: a gated line must never export as a positive training example.
|
|
26
|
-
*
|
|
27
|
-
* That last sentence is enforced here, by `validateRolloutLine`, not merely
|
|
28
|
-
* documented. Validating `reward` and `realness_gated` independently — each a
|
|
29
|
-
* well-typed field, their COMBINATION unchecked — is what let a line claiming
|
|
30
|
-
* `{reward: 0.95, realness_gated: true}` validate clean and walk into every
|
|
31
|
-
* training export. The relationship between the two IS the invariant, so it is
|
|
32
|
-
* checked where every other structural claim about a line is checked.
|
|
33
|
-
*
|
|
34
|
-
* The invariant is about the OUTCOME, not about one field of it. Zeroing
|
|
35
|
-
* `reward` while `outcome.metrics` still carried the per-layer scores that
|
|
36
|
-
* reward was computed from exported the gamed signal anyway, in the dict the
|
|
37
|
-
* verifiers format reads as its per-rubric scores. So `gateGamedOutcome`
|
|
38
|
-
* transforms the whole outcome once, at `assertMinted` — the funnel every
|
|
39
|
-
* minted line passes — and the reward-bearing components are relocated to
|
|
40
|
-
* `provenance.gated_evidence`, which no exporter projects.
|
|
41
|
-
*
|
|
42
|
-
* WHICH checks each door applies is not decided in this file. `./gate-checks`
|
|
43
|
-
* owns the canonical list and the total per-entry-point policy; the three doors
|
|
44
|
-
* below (`validateRolloutLine`, `assertRewardGate`, `assertMinted`) each call
|
|
45
|
-
* `gateErrors` with their declared policy, so a check added to that list applies
|
|
46
|
-
* here without anyone editing this file, and a check deliberately skipped has to
|
|
47
|
-
* name itself there.
|
|
48
|
-
*/
|
|
49
|
-
declare const ROLLOUT_SCHEMA = "tangle.rollout.v1";
|
|
50
|
-
/** `agent` = a solo evaluation run (no multi-agent topology). */
|
|
51
|
-
type RolloutRole = 'agent' | 'supervisor' | 'worker' | 'proposer' | 'judge' | 'analyst';
|
|
52
|
-
/** Split vocabulary follows `RunRecord.splitTag`, extended with `canary`. */
|
|
53
|
-
type RolloutSplit = 'search' | 'dev' | 'holdout' | 'canary';
|
|
54
|
-
/** 'mint' = joined live from RunRecord + trace by `mintRolloutRows`. */
|
|
55
|
-
type RolloutCapture = 'mint' | 'settle-time' | 'backfill';
|
|
56
|
-
type ChatRole = 'system' | 'user' | 'assistant' | 'tool';
|
|
57
|
-
interface ChatToolCall {
|
|
58
|
-
id: string;
|
|
59
|
-
type: 'function';
|
|
60
|
-
function: {
|
|
61
|
-
name: string;
|
|
62
|
-
/** JSON-encoded argument object, exactly as the model emitted it. */
|
|
63
|
-
arguments: string;
|
|
64
|
-
};
|
|
65
|
-
}
|
|
66
|
-
interface ChatMessage {
|
|
67
|
-
role: ChatRole;
|
|
68
|
-
content: string | null;
|
|
69
|
-
/** Reasoning/thinking channel where the harness captured it (full fidelity). */
|
|
70
|
-
reasoning_content?: string;
|
|
71
|
-
tool_calls?: ChatToolCall[];
|
|
72
|
-
/** Required on role:"tool" — the ChatToolCall this result answers. */
|
|
73
|
-
tool_call_id?: string;
|
|
74
|
-
name?: string;
|
|
75
|
-
/**
|
|
76
|
-
* Harbor ATIF `is_copied_context` (RFC 0001 rule 7): this turn was COPIED IN
|
|
77
|
-
* from another trajectory's context, not produced by the agent on this line.
|
|
78
|
-
* The RFC makes excluding it from SFT a MUST, and `toSftRows` does — training
|
|
79
|
-
* on it teaches the model to author text it never authored, and credits this
|
|
80
|
-
* run for another one's work. Absent = false (authored here).
|
|
81
|
-
*/
|
|
82
|
-
is_copied_context?: boolean;
|
|
83
|
-
}
|
|
84
|
-
interface ToolDef {
|
|
85
|
-
type: 'function';
|
|
86
|
-
function: {
|
|
87
|
-
name: string;
|
|
88
|
-
description?: string;
|
|
89
|
-
parameters?: Record<string, unknown>;
|
|
90
|
-
};
|
|
91
|
-
}
|
|
92
|
-
/**
|
|
93
|
-
* Compact trace-span projection (llm/tool step) carried alongside the
|
|
94
|
-
* conversation when the line was minted from a trace. Optional: lines
|
|
95
|
-
* recovered from harness stores have no span structure.
|
|
96
|
-
*/
|
|
97
|
-
interface RolloutStep {
|
|
98
|
-
kind: string;
|
|
99
|
-
name: string;
|
|
100
|
-
/** llm: last-message summary · tool: stringified args. Scrubbed. */
|
|
101
|
-
input?: string;
|
|
102
|
-
/** llm: output text · tool: stringified result. Scrubbed. */
|
|
103
|
-
output?: string;
|
|
104
|
-
status?: 'ok' | 'error';
|
|
105
|
-
durationMs?: number;
|
|
106
|
-
/**
|
|
107
|
-
* LLM inferences this span represents. 0 = deterministic dispatch with no
|
|
108
|
-
* model call — distinct from absent, which means the producer did not track it.
|
|
109
|
-
*/
|
|
110
|
-
llm_call_count?: number;
|
|
111
|
-
/** Exact prompt tokenization. Removes the ambiguity of re-tokenizing text at train time. */
|
|
112
|
-
prompt_token_ids?: number[];
|
|
113
|
-
/** Exact completion tokenization; aligns index-wise with `logprobs`. */
|
|
114
|
-
completion_token_ids?: number[];
|
|
115
|
-
/**
|
|
116
|
-
* Per-completion-token log probabilities under the sampling policy. Required
|
|
117
|
-
* for off-policy correction (importance weighting) when the rollout was
|
|
118
|
-
* generated by a policy other than the one being trained.
|
|
119
|
-
*/
|
|
120
|
-
logprobs?: number[];
|
|
121
|
-
}
|
|
122
|
-
interface RolloutTask {
|
|
123
|
-
/** Benchmark/suite id (e.g. "swe-bench-verified") or the experiment id. */
|
|
124
|
-
suite: string;
|
|
125
|
-
instance_id: string;
|
|
126
|
-
split: RolloutSplit;
|
|
127
|
-
/** Sampling seed the campaign pinned; null = not recorded. */
|
|
128
|
-
seed: number | null;
|
|
129
|
-
/** Replicate index (0-based). */
|
|
130
|
-
rep: number;
|
|
131
|
-
}
|
|
132
|
-
interface RolloutPolicy {
|
|
133
|
-
/** Harness that drove the invocation (e.g. "opencode", "claude", "pi-loops"). */
|
|
134
|
-
harness: string | null;
|
|
135
|
-
harness_version: string | null;
|
|
136
|
-
model: string | null;
|
|
137
|
-
provider: string | null;
|
|
138
|
-
/** Commit of the agent profile / candidate under evaluation. */
|
|
139
|
-
profile_commit: string | null;
|
|
140
|
-
/** sha256 of the effective prompt (post-steering), when recorded. */
|
|
141
|
-
prompt_hash?: string | null;
|
|
142
|
-
/** sha256 of the effective run config, when recorded. */
|
|
143
|
-
config_hash?: string | null;
|
|
144
|
-
/** Canonical agent-profile cell identity, when the run carries one. */
|
|
145
|
-
agent_profile_cell_id?: string | null;
|
|
146
|
-
/** Sampling params (temperature, top_p, max_tokens…); null = not recorded. */
|
|
147
|
-
sampling: Record<string, unknown> | null;
|
|
148
|
-
}
|
|
149
|
-
interface RolloutOutcome {
|
|
150
|
-
/**
|
|
151
|
-
* THE single scalar training signal — the official verdict.
|
|
152
|
-
* null = no verdict exists for this invocation (a labeled gap, never 0).
|
|
153
|
-
*/
|
|
154
|
-
reward: number | null;
|
|
155
|
-
/** Where the reward came from (judge id; "/inherited" = parent episode's). */
|
|
156
|
-
reward_source: string | null;
|
|
157
|
-
/** Raw judge verdict record, verbatim. */
|
|
158
|
-
verdict: unknown;
|
|
159
|
-
/** Everything that is NOT the scalar reward. */
|
|
160
|
-
metrics: Record<string, unknown>;
|
|
161
|
-
is_completed: boolean;
|
|
162
|
-
is_truncated: boolean;
|
|
163
|
-
error: string | null;
|
|
164
|
-
/**
|
|
165
|
-
* Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run faked
|
|
166
|
-
* its success signal. `true` requires `reward` to be 0 or null — the
|
|
167
|
-
* validator rejects the line otherwise — and the line never qualifies for
|
|
168
|
-
* SFT. Required on the wire: a line that does not state the flag does not
|
|
169
|
-
* validate, so no producer can dodge the gate by omitting it.
|
|
170
|
-
*
|
|
171
|
-
* `true` ALSO requires `metrics` to be empty and `verdict` to be null: the
|
|
172
|
-
* numbers the reward was computed from are relocated to
|
|
173
|
-
* `provenance.gated_evidence` by `gateGamedOutcome`. See that function for
|
|
174
|
-
* why zeroing the scalar alone was not enough.
|
|
175
|
-
*/
|
|
176
|
-
realness_gated: boolean;
|
|
177
|
-
/**
|
|
178
|
-
* Whether an authenticity SCREEN ever RAN on this reward — a different claim
|
|
179
|
-
* from `realness_gated`, which is the screen's VERDICT.
|
|
180
|
-
*
|
|
181
|
-
* `realness_gated: false` reads as "we looked and nothing fired". A producer
|
|
182
|
-
* with no screen at all was emitting exactly that, so a never-screened reward
|
|
183
|
-
* was indistinguishable on the wire from a screened-clean one, and the whole
|
|
184
|
-
* anti-Goodhart apparatus silently treated the first as the second. The two
|
|
185
|
-
* claims are now separable:
|
|
186
|
-
*
|
|
187
|
-
* - `true` — a screen ran; `realness_gated` is its verdict.
|
|
188
|
-
* - `false` — the producer declares it HAS no screen (`unscreenedRewardFields`).
|
|
189
|
-
* `assertMinted` REFUSES such a line when its reward is above
|
|
190
|
-
* zero: an unscreened positive reward is precisely the signal
|
|
191
|
-
* the gate exists to qualify, and nothing has qualified it.
|
|
192
|
-
* - absent — not stated. Pre-unification ledgers land here, as does a
|
|
193
|
-
* `RunRecord` carrying no `outcome.realness` at all. Absent is
|
|
194
|
-
* read as "unknown", never as `false` (which would refuse most
|
|
195
|
-
* of the existing corpus) and never as `true`.
|
|
196
|
-
*/
|
|
197
|
-
realness_screened?: boolean;
|
|
198
|
-
}
|
|
199
|
-
interface RolloutCostBlock {
|
|
200
|
-
usd: number | null;
|
|
201
|
-
tokens_in: number | null;
|
|
202
|
-
tokens_out: number | null;
|
|
203
|
-
tokens_reasoning: number | null;
|
|
204
|
-
cache_read: number | null;
|
|
205
|
-
cache_write: number | null;
|
|
206
|
-
wall_s: number | null;
|
|
207
|
-
/**
|
|
208
|
-
* Total LLM inferences across the invocation (ATIF `llm_call_count`,
|
|
209
|
-
* aggregated). Optional and additive: absent = not tracked, never 0.
|
|
210
|
-
*/
|
|
211
|
-
llm_call_count?: number | null;
|
|
212
|
-
}
|
|
213
|
-
interface RolloutArtifacts {
|
|
214
|
-
patch_path: string | null;
|
|
215
|
-
run_dir: string | null;
|
|
216
|
-
/** Source-of-truth transcript pointer (session id / jsonl path) for audit. */
|
|
217
|
-
transcript_ref: string | null;
|
|
218
|
-
}
|
|
219
|
-
/**
|
|
220
|
-
* The reward-bearing half of a GATED line's outcome, moved off `outcome` and
|
|
221
|
-
* parked here verbatim. Diagnostics, never training input — see
|
|
222
|
-
* `gateGamedOutcome`.
|
|
223
|
-
*/
|
|
224
|
-
interface GatedEvidence {
|
|
225
|
-
/** `outcome.metrics` exactly as the producer measured it. */
|
|
226
|
-
metrics?: Record<string, unknown>;
|
|
227
|
-
/** `outcome.verdict` verbatim — the judge record that claimed the success. */
|
|
228
|
-
verdict?: unknown;
|
|
229
|
-
/**
|
|
230
|
-
* The per-step fields `tangle.rollout.v1` does not declare, parked here when
|
|
231
|
-
* the gate projected `steps[]` down to the schema's own key set.
|
|
232
|
-
*
|
|
233
|
-
* A per-step reward is training signal exactly like the scalar, and `steps`
|
|
234
|
-
* rides through `toRewardRows` verbatim — so a gated line was shipping its
|
|
235
|
-
* step-level credit assignment at full value beside a `reward` of 0.
|
|
236
|
-
*/
|
|
237
|
-
steps?: unknown;
|
|
238
|
-
}
|
|
239
|
-
interface RolloutProvenance {
|
|
240
|
-
captured_at: string;
|
|
241
|
-
capture: RolloutCapture;
|
|
242
|
-
/**
|
|
243
|
-
* Why this line is incomplete. Required when `messages` is empty (the
|
|
244
|
-
* transcript could not be recovered); also set by interchange importers to
|
|
245
|
-
* name a MISSING LABEL — an imported trajectory carries no verdict, so
|
|
246
|
-
* `outcome.reward` is null and this says why.
|
|
247
|
-
*/
|
|
248
|
-
gap?: string;
|
|
249
|
-
/**
|
|
250
|
-
* Present only on a realness-gated line: the outcome fields the gate
|
|
251
|
-
* relocated, kept so an auditor can still see WHY the run was gated and what
|
|
252
|
-
* it claimed. Deliberately OUTSIDE `outcome`, because every training exporter
|
|
253
|
-
* reads `outcome` and none reads `provenance`.
|
|
254
|
-
*/
|
|
255
|
-
gated_evidence?: GatedEvidence;
|
|
256
|
-
}
|
|
257
|
-
interface RolloutLine {
|
|
258
|
-
schema: typeof ROLLOUT_SCHEMA;
|
|
259
|
-
rollout_id: string;
|
|
260
|
-
/** Spawning invocation within the same episode (worker → supervisor). */
|
|
261
|
-
parent_rollout_id: string | null;
|
|
262
|
-
run_id: string;
|
|
263
|
-
/** Logical experiment grouping from `RunRecord.experimentId`; null = not recorded. */
|
|
264
|
-
experiment_id: string | null;
|
|
265
|
-
/** Stable candidate identity from `RunRecord.candidateId`; null = not recorded. */
|
|
266
|
-
candidate_id: string | null;
|
|
267
|
-
/** Improvement-loop generation (-1 = baseline); null = not an improvement loop. */
|
|
268
|
-
generation: number | null;
|
|
269
|
-
/** Improvement-loop candidate index (-1 = baseline); null = not an improvement loop. */
|
|
270
|
-
candidate_index: number | null;
|
|
271
|
-
role: RolloutRole;
|
|
272
|
-
task: RolloutTask;
|
|
273
|
-
policy: RolloutPolicy;
|
|
274
|
-
/** Full transcript, inline. [] = gap line (see provenance.gap). */
|
|
275
|
-
messages: ChatMessage[];
|
|
276
|
-
tool_defs: ToolDef[];
|
|
277
|
-
/** Trace-span projections, when minted from a trace. */
|
|
278
|
-
steps?: RolloutStep[];
|
|
279
|
-
outcome: RolloutOutcome;
|
|
280
|
-
cost: RolloutCostBlock;
|
|
281
|
-
artifacts: RolloutArtifacts;
|
|
282
|
-
provenance: RolloutProvenance;
|
|
283
|
-
}
|
|
284
|
-
|
|
285
|
-
/**
|
|
286
|
-
* Supervisor-run analysis — the multi-agent analogue of single-rollout trace
|
|
287
|
-
* analysis. A solo rollout is one invocation with a transcript; a supervisor
|
|
288
|
-
* run is a TREE of invocations (a brain that spawns, steers, and settles
|
|
289
|
-
* workers) plus the event timeline that connects them. `src/trace-analyst`
|
|
290
|
-
* answers "what happened inside one session"; this module answers "what did
|
|
291
|
-
* the tree do" — did the brain steer anyone mid-task, how many spawn waves,
|
|
292
|
-
* how concurrent, how idle, what did each role cost, what came back.
|
|
293
|
-
*
|
|
294
|
-
* The nodes of that tree are NOT a new shape: they are `tangle.rollout.v1`
|
|
295
|
-
* rows (`src/rollout`), keyed by `parent_rollout_id`, with `role` already
|
|
296
|
-
* spanning `supervisor` / `worker`. `supervisorRunRolloutLines` mints them.
|
|
297
|
-
* What rollout rows deliberately do NOT carry is the inter-invocation event
|
|
298
|
-
* timeline (spawn/settle/steer instants), which is what every structural
|
|
299
|
-
* metric here is computed from — so the reader consumes the journal event
|
|
300
|
-
* stream and emits rollout rows, rather than maintaining a parallel node type.
|
|
301
|
-
*
|
|
302
|
-
* ## UNAVAILABLE ≠ ZERO
|
|
303
|
-
*
|
|
304
|
-
* Every metric whose backing artifact can be missing is typed
|
|
305
|
-
* `Measured<T> = T | { unavailable: reason }`. A supervisor that steered
|
|
306
|
-
* nobody reports `steers: 0`; a supervisor whose worker logs were never
|
|
307
|
-
* written reports `steers: unavailable — <reason>`. The two have driven
|
|
308
|
-
* opposite conclusions about the same architecture, so they never collapse.
|
|
309
|
-
*/
|
|
310
|
-
|
|
311
|
-
/** A metric that could not be computed, with the reason its artifact was missing. */
|
|
312
|
-
interface Unavailable {
|
|
313
|
-
readonly unavailable: string;
|
|
314
|
-
}
|
|
315
|
-
/** A metric value, or the reason it is unknown. NEVER collapse `unavailable` to 0. */
|
|
316
|
-
type Measured<T> = T | Unavailable;
|
|
317
|
-
declare function unavailable(reason: string): Unavailable;
|
|
318
|
-
declare function isUnavailable(v: unknown): v is Unavailable;
|
|
319
|
-
/** Render a measured scalar for the markdown/headline: `0` and `unavailable` stay distinct. */
|
|
320
|
-
declare function showMeasured(v: Measured<number | string | boolean | null>): string;
|
|
321
|
-
/** One worker's logs, as read. `null` = the artifact did not exist. */
|
|
322
|
-
interface WorkerLogSource {
|
|
323
|
-
readonly label: string;
|
|
324
|
-
/** Worker event stream — started / progress / finished / message events (JSONL). */
|
|
325
|
-
readonly events: string | null;
|
|
326
|
-
/** The durable steer queue — one line per steer request (JSONL). */
|
|
327
|
-
readonly inbox: string | null;
|
|
328
|
-
/** Worker patch byte length, or null when absent. */
|
|
329
|
-
readonly patchBytes: number | null;
|
|
330
|
-
/** Where this worker's transcript lives, for the rollout row. Null = no such artifact. */
|
|
331
|
-
readonly transcriptRef?: string | null;
|
|
332
|
-
/** Where this worker's delivered patch lives. Null = the store keeps no patch per worker. */
|
|
333
|
-
readonly patchPath?: string | null;
|
|
334
|
-
/** This worker's own inference tokens, when the store records them per worker. */
|
|
335
|
-
readonly tokensIn?: number | null;
|
|
336
|
-
readonly tokensOut?: number | null;
|
|
337
|
-
readonly cacheRead?: number | null;
|
|
338
|
-
readonly cacheWrite?: number | null;
|
|
339
|
-
}
|
|
340
|
-
/**
|
|
341
|
-
* Facts a SOURCE structurally cannot express, each with the reason.
|
|
342
|
-
*
|
|
343
|
-
* The difference between "the artifact is missing" and "this store never
|
|
344
|
-
* records that fact" is the difference between a run that spent $0 and a
|
|
345
|
-
* harness that does not price inference — and the second harness is where a
|
|
346
|
-
* loops-shaped assumption becomes a fabricated zero. A reader declares its
|
|
347
|
-
* limits once; the analyzer reports `unavailable` for everything downstream.
|
|
348
|
-
*
|
|
349
|
-
* `null` on a field means the source DOES carry that fact.
|
|
350
|
-
*/
|
|
351
|
-
interface SourceLimits {
|
|
352
|
-
/** Reason inference spend has no price in this store (null = the store prices it). */
|
|
353
|
-
readonly spendUsd: string | null;
|
|
354
|
-
/** Reason workers carry no pass/fail verdict (null = verdicts are recorded). */
|
|
355
|
-
readonly workerVerdicts: string | null;
|
|
356
|
-
/** Reason no delivered artifact (patch/diff) is retained per worker (null = retained). */
|
|
357
|
-
readonly deliverables: string | null;
|
|
358
|
-
}
|
|
359
|
-
/** A source that carries every fact the analyzer can use. */
|
|
360
|
-
declare const NO_SOURCE_LIMITS: SourceLimits;
|
|
361
|
-
/**
|
|
362
|
-
* Everything the pure analyzer reads — already-read bytes, never paths. Each
|
|
363
|
-
* field is `null` when its artifact was absent, which is what turns the
|
|
364
|
-
* dependent metrics into `unavailable` rather than 0.
|
|
365
|
-
*
|
|
366
|
-
* This is the whole input contract. Any store that can produce these strings
|
|
367
|
-
* (an on-disk loops run, an object-store archive, a database, a test fixture)
|
|
368
|
-
* is a valid source; `loopsSupervisorRunReader` is ONE implementation.
|
|
369
|
-
*/
|
|
370
|
-
interface SupervisorRunSources {
|
|
371
|
-
/** Stable identity of the run being analyzed (a directory, a run id, a URL). */
|
|
372
|
-
readonly runRef: string;
|
|
373
|
-
readonly instanceId: string | null;
|
|
374
|
-
/** Which arm/variant of a comparison this run is, when the run belongs to one. */
|
|
375
|
-
readonly arm: string | null;
|
|
376
|
-
/** Identity of the supervision-tree store this was read from; null = none found. */
|
|
377
|
-
readonly supRunDir: string | null;
|
|
378
|
-
/** Supervision journal — spawned / settled / cancelled / metered events (JSONL). */
|
|
379
|
-
readonly journal: string | null;
|
|
380
|
-
/** Per-brain-call tap (JSONL): finish_reason, completion tokens, requested max tokens. */
|
|
381
|
-
readonly brainLog: string | null;
|
|
382
|
-
/** Supervisor state document (JSON). */
|
|
383
|
-
readonly state: string | null;
|
|
384
|
-
/** Supervisor progress stream (JSONL). */
|
|
385
|
-
readonly progress: string | null;
|
|
386
|
-
/** Per-worker logs; `null` = the worker log store itself was missing. */
|
|
387
|
-
readonly workers: readonly WorkerLogSource[] | null;
|
|
388
|
-
/** Why `workers` is null (only set when it is). */
|
|
389
|
-
readonly workersMissingReason: string | null;
|
|
390
|
-
/** Run result document (JSON). */
|
|
391
|
-
readonly result: string | null;
|
|
392
|
-
/**
|
|
393
|
-
* Judge verdict document (JSON), or the matching ledger row re-encoded as
|
|
394
|
-
* one. Runners that write the verdict straight to a ledger leave no judge
|
|
395
|
-
* document, so the ledger row is the same fact from the same run — not a
|
|
396
|
-
* substitute measurement.
|
|
397
|
-
*/
|
|
398
|
-
readonly judge: string | null;
|
|
399
|
-
/** Where `judge` came from, for the report's provenance line. */
|
|
400
|
-
readonly judgeSource: string | null;
|
|
401
|
-
/** Delivered unified-diff patch text. */
|
|
402
|
-
readonly patch: string | null;
|
|
403
|
-
/** Outer-driver log (used for the driver's steer verbs + deadline evidence). */
|
|
404
|
-
readonly driverLog: string | null;
|
|
405
|
-
/**
|
|
406
|
-
* Worker tokens recovered from a harness session store; null = store unavailable.
|
|
407
|
-
* `store` names the store in the report's provenance line (e.g. `opencode`).
|
|
408
|
-
*/
|
|
409
|
-
readonly harnessWorkerTokens: {
|
|
410
|
-
store: string;
|
|
411
|
-
sessions: number;
|
|
412
|
-
input: number;
|
|
413
|
-
output: number;
|
|
414
|
-
/** Cached prompt tokens, when the store counts them separately. */
|
|
415
|
-
cacheRead?: number;
|
|
416
|
-
cacheWrite?: number;
|
|
417
|
-
} | null;
|
|
418
|
-
readonly harnessMissingReason: string | null;
|
|
419
|
-
/** What this store structurally cannot record. See `SourceLimits`. */
|
|
420
|
-
readonly limits: SourceLimits;
|
|
421
|
-
/**
|
|
422
|
-
* Where the ROOT invocation's transcript lives. Undefined lets the rollout
|
|
423
|
-
* minter fall back to the loops layout (`<supRunDir>/journal.jsonl`); any
|
|
424
|
-
* other store must say, or the row points at a path that never existed.
|
|
425
|
-
*/
|
|
426
|
-
readonly rootTranscriptRef?: string | null;
|
|
427
|
-
/**
|
|
428
|
-
* The `traces` CLI command that covers this run's harness-session layer.
|
|
429
|
-
* Null falls back to the analyzer's default (an opencode worker fleet).
|
|
430
|
-
*/
|
|
431
|
-
readonly traceCommand: string | null;
|
|
432
|
-
}
|
|
433
|
-
/**
|
|
434
|
-
* A source of supervisor-run bytes. Implementations own their storage layout;
|
|
435
|
-
* the analyzer only ever sees `SupervisorRunSources`.
|
|
436
|
-
*/
|
|
437
|
-
interface SupervisorRunReader {
|
|
438
|
-
/** Stable identity of what this reader points at (for logs and report labels). */
|
|
439
|
-
readonly runRef: string;
|
|
440
|
-
read(): Promise<SupervisorRunSources>;
|
|
441
|
-
}
|
|
442
|
-
declare const SUPERVISOR_RUN_SCHEMA = "tangle.supervisor-run@1";
|
|
443
|
-
declare const SUPERVISOR_RUN_ROLLUP_SCHEMA = "tangle.supervisor-run-rollup@1";
|
|
444
|
-
interface SteerBreakdown {
|
|
445
|
-
readonly worker: string;
|
|
446
|
-
/** Steer requests durably queued to this worker's inbox. */
|
|
447
|
-
readonly queued: number;
|
|
448
|
-
/** Steers the worker's executor actually accepted (control event `delivered:true`). */
|
|
449
|
-
readonly delivered: number;
|
|
450
|
-
}
|
|
451
|
-
interface OrchestrationMetrics {
|
|
452
|
-
readonly workersSpawned: Measured<number>;
|
|
453
|
-
readonly workersSettled: Measured<number>;
|
|
454
|
-
readonly workersCancelled: Measured<number>;
|
|
455
|
-
/** THE HEADLINE: mid-task steers the brain sent to live workers. 0 ≠ unavailable. */
|
|
456
|
-
readonly steers: Measured<number>;
|
|
457
|
-
readonly steersDelivered: Measured<number>;
|
|
458
|
-
readonly steersByWorker: Measured<readonly SteerBreakdown[]>;
|
|
459
|
-
/** Outer-driver `supervisor_steer` tool calls seen in the driver log (a second steer path). */
|
|
460
|
-
readonly driverSteerCalls: Measured<number>;
|
|
461
|
-
/**
|
|
462
|
-
* Spawn waves. A wave is a maximal run of worker spawns with no settle/cancel between
|
|
463
|
-
* them: wave N+1 begins at the first spawn issued after at least one worker from an
|
|
464
|
-
* earlier wave has settled. Structural, not a time threshold — no tunable constant.
|
|
465
|
-
*/
|
|
466
|
-
readonly waves: Measured<number>;
|
|
467
|
-
readonly waveSizes: Measured<readonly number[]>;
|
|
468
|
-
readonly maxConcurrency: Measured<number>;
|
|
469
|
-
/** Worker spawns issued after the first settlement — the retry/respawn tail. */
|
|
470
|
-
readonly respawns: Measured<number>;
|
|
471
|
-
/** Labels spawned more than once (a literal retry of the same subtask). */
|
|
472
|
-
readonly repeatedLabels: Measured<readonly string[]>;
|
|
473
|
-
/** Longest parent chain below the root, in worker hops. */
|
|
474
|
-
readonly delegationDepth: Measured<number>;
|
|
475
|
-
readonly timeToFirstSpawnMs: Measured<number>;
|
|
476
|
-
readonly supervisorWallMs: Measured<number>;
|
|
477
|
-
/** Wall time inside the supervisor run with ZERO live workers. */
|
|
478
|
-
readonly idleMs: Measured<number>;
|
|
479
|
-
readonly idlePct: Measured<number>;
|
|
480
|
-
/** sum(worker wall) / supervisor wall. >1 means real parallelism. */
|
|
481
|
-
readonly workerUtilization: Measured<number>;
|
|
482
|
-
}
|
|
483
|
-
interface DecisionMetrics {
|
|
484
|
-
readonly settledByStatus: Measured<Record<string, number>>;
|
|
485
|
-
readonly settledVerdicts: Measured<Record<string, number>>;
|
|
486
|
-
/** Worker verified its own work green AND produced a patch. */
|
|
487
|
-
readonly accepted: Measured<number>;
|
|
488
|
-
/** Worker settled with a failing verify. */
|
|
489
|
-
readonly rejected: Measured<number>;
|
|
490
|
-
/** Worker verified green but delivered no patch bytes — output with nothing to accept. */
|
|
491
|
-
readonly emptyPass: Measured<number>;
|
|
492
|
-
/** Settlements the brain observed before issuing its next spawn (evidence→respawn). */
|
|
493
|
-
readonly observeThenRespawn: Measured<number>;
|
|
494
|
-
/** Respawns with no settled evidence in front of them. */
|
|
495
|
-
readonly respawnWithoutEvidence: Measured<number>;
|
|
496
|
-
/** Steer + question traffic on the live down/up legs — the only "review while running" signal. */
|
|
497
|
-
readonly reviewActions: Measured<number>;
|
|
498
|
-
readonly workerEvidenceBytes: Measured<number>;
|
|
499
|
-
}
|
|
500
|
-
interface RoleSpend {
|
|
501
|
-
readonly tokensIn: Measured<number>;
|
|
502
|
-
readonly tokensOut: Measured<number>;
|
|
503
|
-
/**
|
|
504
|
-
* Cached prompt tokens read/written. On a harness that caches aggressively
|
|
505
|
-
* these dwarf `tokensIn`, so a report that omits them understates the context
|
|
506
|
-
* each invocation actually consumed. `unavailable` = the store has no such counter.
|
|
507
|
-
*/
|
|
508
|
-
readonly cacheRead: Measured<number>;
|
|
509
|
-
readonly cacheWrite: Measured<number>;
|
|
510
|
-
readonly usd: Measured<number>;
|
|
511
|
-
readonly source: string;
|
|
512
|
-
}
|
|
513
|
-
interface PerWorkerRow {
|
|
514
|
-
readonly worker: string;
|
|
515
|
-
readonly wallMs: number | null;
|
|
516
|
-
/** `null` = this store does not attribute tokens per worker (NOT "zero tokens"). */
|
|
517
|
-
readonly tokensIn: number | null;
|
|
518
|
-
readonly tokensOut: number | null;
|
|
519
|
-
readonly usd: number | null;
|
|
520
|
-
readonly patchBytes: number | null;
|
|
521
|
-
readonly passed: boolean | null;
|
|
522
|
-
}
|
|
523
|
-
interface WallDistribution {
|
|
524
|
-
readonly n: number;
|
|
525
|
-
readonly min: number;
|
|
526
|
-
readonly p50: number;
|
|
527
|
-
readonly p90: number;
|
|
528
|
-
readonly max: number;
|
|
529
|
-
readonly sum: number;
|
|
530
|
-
}
|
|
531
|
-
interface EconomicsMetrics {
|
|
532
|
-
/** Driver/brain inference — journal `metered` events. */
|
|
533
|
-
readonly brain: RoleSpend;
|
|
534
|
-
/**
|
|
535
|
-
* Brain completions that came back `finish_reason: "length"` — output TRUNCATED. Any value
|
|
536
|
-
* above 0 means the supervisor planned into a wall and then acted on the half-written plan,
|
|
537
|
-
* which is a defect and not a cost figure. The journal's `metered` rows carry token counts
|
|
538
|
-
* but no finish reason, so this reads the per-call brain tap; a run whose supervisor
|
|
539
|
-
* predates that tap reports `unavailable`, never 0.
|
|
540
|
-
*/
|
|
541
|
-
readonly brainTruncations: Measured<number>;
|
|
542
|
-
/** Worker inference — journal `settled` spend plus the harness session join. */
|
|
543
|
-
readonly workers: RoleSpend;
|
|
544
|
-
readonly totalUsd: Measured<number>;
|
|
545
|
-
/**
|
|
546
|
-
* Where `totalUsd` came from. CLI-backend workers never price their own inference into
|
|
547
|
-
* the journal, so on those arms the total is BRAIN-ONLY and the worker row's token
|
|
548
|
-
* counts (recovered from the harness store) are the honest worker-side figure.
|
|
549
|
-
*/
|
|
550
|
-
readonly totalUsdSource: string;
|
|
551
|
-
readonly costPerAcceptedPatchUsd: Measured<number>;
|
|
552
|
-
readonly workerWallMsDistribution: Measured<WallDistribution>;
|
|
553
|
-
readonly perWorker: Measured<readonly PerWorkerRow[]>;
|
|
554
|
-
}
|
|
555
|
-
interface PatchStats {
|
|
556
|
-
readonly files: number;
|
|
557
|
-
readonly linesAdded: number;
|
|
558
|
-
readonly linesRemoved: number;
|
|
559
|
-
readonly testFilesTouched: readonly string[];
|
|
560
|
-
}
|
|
561
|
-
interface OutcomeMetrics {
|
|
562
|
-
readonly supStatus: Measured<string>;
|
|
563
|
-
readonly supVerdict: Measured<string>;
|
|
564
|
-
readonly delivered: Measured<boolean>;
|
|
565
|
-
readonly judgeResolved: Measured<boolean | null>;
|
|
566
|
-
readonly judgeScore: Measured<number | null>;
|
|
567
|
-
readonly judgePassed: Measured<number | null>;
|
|
568
|
-
readonly judgeTotal: Measured<number | null>;
|
|
569
|
-
readonly verifyPass: Measured<boolean>;
|
|
570
|
-
readonly verifyRc: Measured<number>;
|
|
571
|
-
readonly patch: Measured<PatchStats>;
|
|
572
|
-
/** Which document the judge fields came from (a judge file, a ledger row, or nothing). */
|
|
573
|
-
readonly judgeSource: string | null;
|
|
574
|
-
}
|
|
575
|
-
interface SupervisorRunReport {
|
|
576
|
-
readonly schema: typeof SUPERVISOR_RUN_SCHEMA;
|
|
577
|
-
/** The `runRef` of the sources this report was computed from. */
|
|
578
|
-
readonly runRef: string;
|
|
579
|
-
readonly instanceId: string | null;
|
|
580
|
-
readonly arm: string | null;
|
|
581
|
-
readonly supervisorId: Measured<string>;
|
|
582
|
-
readonly generatedAt: string;
|
|
583
|
-
readonly orchestration: OrchestrationMetrics;
|
|
584
|
-
readonly decision: DecisionMetrics;
|
|
585
|
-
readonly economics: EconomicsMetrics;
|
|
586
|
-
readonly outcome: OutcomeMetrics;
|
|
587
|
-
/** Artifacts that were missing, in read order — the provenance of every `unavailable`. */
|
|
588
|
-
readonly gaps: readonly string[];
|
|
589
|
-
/** The `traces` CLI command that covers the harness-session layer for this run. */
|
|
590
|
-
readonly traceCommand: string;
|
|
591
|
-
}
|
|
592
|
-
interface RollupCellRow {
|
|
593
|
-
readonly instanceId: string | null;
|
|
594
|
-
readonly arm: string | null;
|
|
595
|
-
readonly steers: Measured<number>;
|
|
596
|
-
readonly waves: Measured<number>;
|
|
597
|
-
readonly utilization: Measured<number>;
|
|
598
|
-
readonly idlePct: Measured<number>;
|
|
599
|
-
readonly resolved: Measured<boolean | null>;
|
|
600
|
-
readonly usd: Measured<number>;
|
|
601
|
-
}
|
|
602
|
-
interface SupervisorRunRollup {
|
|
603
|
-
readonly schema: typeof SUPERVISOR_RUN_ROLLUP_SCHEMA;
|
|
604
|
-
readonly cells: number;
|
|
605
|
-
readonly steersTotal: Measured<number>;
|
|
606
|
-
readonly cellsWithSteers: Measured<number>;
|
|
607
|
-
readonly cellsWithUnavailableSteers: number;
|
|
608
|
-
readonly wavesMean: Measured<number>;
|
|
609
|
-
readonly maxConcurrencyMax: Measured<number>;
|
|
610
|
-
readonly utilizationMean: Measured<number>;
|
|
611
|
-
readonly idlePctMean: Measured<number>;
|
|
612
|
-
readonly workersSpawnedTotal: Measured<number>;
|
|
613
|
-
readonly acceptedTotal: Measured<number>;
|
|
614
|
-
readonly usdTotal: Measured<number>;
|
|
615
|
-
readonly resolvedCount: Measured<number>;
|
|
616
|
-
readonly perCell: readonly RollupCellRow[];
|
|
617
|
-
}
|
|
618
|
-
/**
|
|
619
|
-
* A supervision tree expressed in the canonical rollout row type: one
|
|
620
|
-
* `RolloutLine` per invocation, joined by `parent_rollout_id`. The root row
|
|
621
|
-
* carries `role: 'supervisor'`; every spawned worker carries `role: 'worker'`
|
|
622
|
-
* with the root as its parent.
|
|
623
|
-
*/
|
|
624
|
-
interface SupervisorRunTree {
|
|
625
|
-
readonly rootId: string | null;
|
|
626
|
-
readonly nodes: readonly RolloutLine[];
|
|
627
|
-
/** Why a node could not be recovered, in read order. */
|
|
628
|
-
readonly gaps: readonly string[];
|
|
629
|
-
}
|
|
630
|
-
|
|
631
|
-
/**
|
|
632
|
-
* The pure analyzer. Takes already-read bytes (`SupervisorRunSources`) and
|
|
633
|
-
* returns the report — every metric derivable from a synthetic journal string
|
|
634
|
-
* with no filesystem, no process, and no network. All I/O lives in a reader
|
|
635
|
-
* (`loops-reader.ts` is one).
|
|
636
|
-
*/
|
|
637
|
-
|
|
638
|
-
interface Tokens {
|
|
639
|
-
input: number;
|
|
640
|
-
output: number;
|
|
641
|
-
cacheRead: number;
|
|
642
|
-
cacheWrite: number;
|
|
643
|
-
/** False when the event carried no cache counters at all — not "zero cached". */
|
|
644
|
-
hasCache: boolean;
|
|
645
|
-
}
|
|
646
|
-
interface SpendLike {
|
|
647
|
-
tokens: Tokens;
|
|
648
|
-
usd: number;
|
|
649
|
-
}
|
|
650
|
-
interface SpawnRow {
|
|
651
|
-
id: string;
|
|
652
|
-
parent: string | null;
|
|
653
|
-
label: string;
|
|
654
|
-
at: number | null;
|
|
655
|
-
}
|
|
656
|
-
interface CloseRow {
|
|
657
|
-
id: string;
|
|
658
|
-
kind: 'settled' | 'cancelled';
|
|
659
|
-
status: string | null;
|
|
660
|
-
verdict: string | null;
|
|
661
|
-
at: number | null;
|
|
662
|
-
spend: SpendLike;
|
|
663
|
-
/** False when the close event carried no spend object — not "spent nothing". */
|
|
664
|
-
hasSpend: boolean;
|
|
665
|
-
}
|
|
666
|
-
interface WorkerLogFacts {
|
|
667
|
-
started: number | null;
|
|
668
|
-
/** True once a `finished` event was seen — independent of whether its `at` parsed. */
|
|
669
|
-
finished: boolean;
|
|
670
|
-
finishedAt: number | null;
|
|
671
|
-
passed: boolean | null;
|
|
672
|
-
/** `patchBytes` as reported by the finished event (not the patch file's size). */
|
|
673
|
-
finishedPatchBytes: number | null;
|
|
674
|
-
evidenceBytes: number;
|
|
675
|
-
steersQueued: number;
|
|
676
|
-
steersDelivered: number;
|
|
677
|
-
questions: number;
|
|
678
|
-
}
|
|
679
|
-
/**
|
|
680
|
-
* The tree + timeline the report is computed from, exposed because the rollout-row
|
|
681
|
-
* minter needs exactly the same parse (one parser, two consumers).
|
|
682
|
-
*/
|
|
683
|
-
interface SupervisorTreeFacts {
|
|
684
|
-
readonly rootId: string | null;
|
|
685
|
-
readonly spawns: readonly SpawnRow[];
|
|
686
|
-
readonly closes: readonly CloseRow[];
|
|
687
|
-
readonly workerSpawns: readonly SpawnRow[];
|
|
688
|
-
readonly workerCloses: readonly CloseRow[];
|
|
689
|
-
readonly brain: {
|
|
690
|
-
tokensIn: number;
|
|
691
|
-
tokensOut: number;
|
|
692
|
-
cacheRead: number;
|
|
693
|
-
cacheWrite: number;
|
|
694
|
-
/** False when no metered event carried cache counters — not "nothing cached". */
|
|
695
|
-
hasCache: boolean;
|
|
696
|
-
usd: number;
|
|
697
|
-
meteredCount: number;
|
|
698
|
-
};
|
|
699
|
-
readonly workerLogs: ReadonlyMap<string, WorkerLogFacts>;
|
|
700
|
-
readonly startedAt: number | null;
|
|
701
|
-
readonly completedAt: number | null;
|
|
702
|
-
}
|
|
703
|
-
declare function parseSupervisorTree(src: SupervisorRunSources): SupervisorTreeFacts;
|
|
704
|
-
/**
|
|
705
|
-
* Analyze already-read supervisor-run bytes. Pure and synchronous: same bytes
|
|
706
|
-
* in, same report out (modulo `generatedAt`, which `now` pins in tests).
|
|
707
|
-
*/
|
|
708
|
-
declare function analyzeSupervisorRunSources(src: SupervisorRunSources, now?: () => number): SupervisorRunReport;
|
|
709
|
-
/** Unified-diff stats. Counts `+++ b/<path>` targets, body +/- lines, and test-file touches. */
|
|
710
|
-
declare function parsePatch(text: string): PatchStats;
|
|
711
|
-
/**
|
|
712
|
-
* Aggregate many supervisor-run reports. A metric no run could measure stays
|
|
713
|
-
* `unavailable` rather than becoming a 0-valued mean, and cells whose steer
|
|
714
|
-
* count was unavailable are counted separately from cells that measured zero.
|
|
715
|
-
*/
|
|
716
|
-
declare function rollupSupervisorRuns(reports: readonly SupervisorRunReport[]): SupervisorRunRollup;
|
|
717
|
-
|
|
718
|
-
/**
|
|
719
|
-
* Supervision-tree reader over a THIRD-PARTY harness: Claude Code.
|
|
720
|
-
*
|
|
721
|
-
* `loops-reader.ts` reads a supervisor we wrote, whose journal was designed
|
|
722
|
-
* for this analysis. This reader reads a harness we do not control, whose
|
|
723
|
-
* transcript was designed for replaying a chat — and recovers the same tree
|
|
724
|
-
* from it. If both produce a `SupervisorRunSources`, the tree model is a
|
|
725
|
-
* property of multi-agent runs, not of our journal format.
|
|
726
|
-
*
|
|
727
|
-
* ## Where the tree hides in a Claude Code transcript
|
|
728
|
-
*
|
|
729
|
-
* | Tree fact | Claude Code evidence |
|
|
730
|
-
* |---|---|
|
|
731
|
-
* | spawn | assistant `tool_use` (`Agent` / `Task`), answered by a `tool_result` whose `toolUseResult.agentId` names the child |
|
|
732
|
-
* | settle | a `<task-notification>` block in a later user line: `<task-id>` = agentId, `<status>` |
|
|
733
|
-
* | steer | assistant `tool_use` (`SendMessage`) with `input.to` = agentId — mid-task, to a LIVE child |
|
|
734
|
-
* | delivered | that steer's `tool_result` carrying `success` / `resumedAgentId` |
|
|
735
|
-
* | cancel | assistant `tool_use` (`TaskStop`) targeting an agentId |
|
|
736
|
-
* | brain spend| `message.usage` on the main thread's assistant lines |
|
|
737
|
-
* | worker spend| `message.usage` inside `<session>/subagents/agent-<id>.jsonl` |
|
|
738
|
-
* | depth | a child transcript that itself contains `Agent` tool_use lines |
|
|
739
|
-
*
|
|
740
|
-
* Every one of those is read through `parseClaudeEntries` — the SAME line
|
|
741
|
-
* parser `src/rollout/readers/claude-jsonl.ts` uses for solo rollouts. There
|
|
742
|
-
* is no second transcript parser.
|
|
743
|
-
*
|
|
744
|
-
* ## What Claude Code cannot say
|
|
745
|
-
*
|
|
746
|
-
* It records tokens but never a price, runs no per-worker verify, and keeps no
|
|
747
|
-
* per-worker patch. Those are declared once in `limits`, so the analyzer
|
|
748
|
-
* reports `unavailable — <reason>` instead of the $0 / 0-accepted that summing
|
|
749
|
-
* an empty field would produce. See `SourceLimits`.
|
|
750
|
-
*
|
|
751
|
-
* ## Metric coverage vs the loops journal
|
|
752
|
-
*
|
|
753
|
-
* Measured on a real 52-agent session (fixture:
|
|
754
|
-
* `tests/fixtures/supervisor-run/claude-code-session-*`).
|
|
755
|
-
*
|
|
756
|
-
* | Metric | loops | Claude Code | Why |
|
|
757
|
-
* |---|---|---|---|
|
|
758
|
-
* | workersSpawned / Settled / Cancelled | full | full | spawn tool_use + task-notification + TaskStop |
|
|
759
|
-
* | steers / steersDelivered / steersByWorker | full | full | `SendMessage`; delivery from its tool_result |
|
|
760
|
-
* | waves / waveSizes / maxConcurrency | full | full | derived from spawn/settle instants |
|
|
761
|
-
* | respawns / repeatedLabels | full | full | same derivation |
|
|
762
|
-
* | delegationDepth | full | full | a child transcript's own spawn calls |
|
|
763
|
-
* | timeToFirstSpawn / supervisorWall | full | full | transcript instants |
|
|
764
|
-
* | idleMs / idlePct / workerUtilization | full | PARTIAL | an agent that never notifies is counted live to the end of the transcript |
|
|
765
|
-
* | observeThenRespawn / respawnWithoutEvidence | full | full | ordering of spawn vs settle instants |
|
|
766
|
-
* | workerEvidenceBytes | full | PARTIAL | the child's closing message; 0 for pruned transcripts |
|
|
767
|
-
* | brain tokens in/out + cache | full | full | main-thread `message.usage` |
|
|
768
|
-
* | worker tokens in/out + cache | via harness join | PARTIAL | only for retained subagent transcripts |
|
|
769
|
-
* | perWorker wall | full | full | spawn → settle instants |
|
|
770
|
-
* | accepted / rejected / emptyPass / settledVerdicts | full | NONE | no per-worker verify step exists |
|
|
771
|
-
* | brain/worker/total usd, costPerAcceptedPatch | full | NONE | transcripts carry no price |
|
|
772
|
-
* | patch stats, delivered, verifyPass/Rc | full | NONE | no diff is handed back |
|
|
773
|
-
* | judgeResolved / Score / Passed / Total | full | NONE | no judge in the loop |
|
|
774
|
-
* | driverSteerCalls, brainTruncations | full | NONE | no outer driver log, no per-call finish_reason tap |
|
|
775
|
-
*/
|
|
776
|
-
|
|
777
|
-
/** Tool names that spawn a child agent. `Task` is the older name for `Agent`. */
|
|
778
|
-
declare const DEFAULT_SPAWN_TOOLS: readonly ["Agent", "Task"];
|
|
779
|
-
/** Tool names that deliver a message to an ALREADY-RUNNING child agent. */
|
|
780
|
-
declare const DEFAULT_STEER_TOOLS: readonly ["SendMessage"];
|
|
781
|
-
/** Tool names that stop a running child agent. */
|
|
782
|
-
declare const DEFAULT_CANCEL_TOOLS: readonly ["TaskStop", "KillAgent"];
|
|
783
|
-
interface ClaudeCodeReaderOptions {
|
|
784
|
-
/** The main session transcript: `~/.claude/projects/<slug>/<sessionId>.jsonl`. */
|
|
785
|
-
readonly transcriptPath: string;
|
|
786
|
-
/**
|
|
787
|
-
* Directory of child transcripts. Defaults to `<transcript-dir>/<sessionId>/subagents`.
|
|
788
|
-
* `null` skips the join, and every per-worker token count becomes unavailable.
|
|
789
|
-
*/
|
|
790
|
-
readonly subagentsDir?: string | null;
|
|
791
|
-
readonly runRef?: string;
|
|
792
|
-
readonly instanceId?: string | null;
|
|
793
|
-
readonly arm?: string | null;
|
|
794
|
-
readonly spawnTools?: readonly string[];
|
|
795
|
-
readonly steerTools?: readonly string[];
|
|
796
|
-
readonly cancelTools?: readonly string[];
|
|
797
|
-
}
|
|
798
|
-
/**
|
|
799
|
-
* Read a Claude Code session (plus its subagent transcripts) as supervision-tree
|
|
800
|
-
* source bytes. Never throws on a missing artifact.
|
|
801
|
-
*/
|
|
802
|
-
declare function readClaudeCodeSupervisorRun(opts: ClaudeCodeReaderOptions): Promise<SupervisorRunSources>;
|
|
803
|
-
/** A `SupervisorRunReader` over a Claude Code session — the same contract loops implements. */
|
|
804
|
-
declare function claudeCodeSupervisorRunReader(opts: ClaudeCodeReaderOptions): SupervisorRunReader;
|
|
805
|
-
|
|
806
|
-
/**
|
|
807
|
-
* ONE implementation of `SupervisorRunReader`: the on-disk layout the loops
|
|
808
|
-
* supervisor writes — `<runDir>/ws/.loops/supervisor/<id>/{journal.jsonl,
|
|
809
|
-
* state.json, progress.ndjson, workers/*.ndjson}` alongside the run's
|
|
810
|
-
* `result.json` / `judge.json` / `driver.log` / delivered patch.
|
|
811
|
-
*
|
|
812
|
-
* Nothing in `analyze.ts` knows this layout exists. A different store (an
|
|
813
|
-
* archive, an object bucket, a database) implements the same interface and
|
|
814
|
-
* gets the same report.
|
|
815
|
-
*
|
|
816
|
-
* Worker token recovery reuses the rollout module's opencode reader rather
|
|
817
|
-
* than opening a second sqlite path — one store client, one corruption policy.
|
|
818
|
-
*/
|
|
819
|
-
|
|
820
|
-
/** Locate the (single) supervisor run dir under `<ws>/.loops/supervisor`. */
|
|
821
|
-
declare function findSupervisorRunDirIn(ws: string): Promise<string | null>;
|
|
822
|
-
interface LoopsReaderOptions {
|
|
823
|
-
/** Override the workspace dir (default `<runDir>/ws`). */
|
|
824
|
-
readonly ws?: string;
|
|
825
|
-
/** Delivered patch path (default: `patchPath` from result.json). */
|
|
826
|
-
readonly patchPath?: string;
|
|
827
|
-
/** opencode sqlite store; set to `null` to skip the worker-token join entirely. */
|
|
828
|
-
readonly opencodeDb?: string | null;
|
|
829
|
-
/** Ledger to fall back to when the run has no `judge.json` (matched on iid + arm + runDir). */
|
|
830
|
-
readonly ledgerPath?: string;
|
|
831
|
-
}
|
|
832
|
-
/**
|
|
833
|
-
* Read a loops supervisor run directory into source bytes. Never throws on a
|
|
834
|
-
* missing artifact — an absent file becomes a `null` field, which is what makes
|
|
835
|
-
* the dependent metric `unavailable` instead of 0.
|
|
836
|
-
*/
|
|
837
|
-
declare function readLoopsSupervisorRun(runDir: string, opts?: LoopsReaderOptions): Promise<SupervisorRunSources>;
|
|
838
|
-
/** The loops on-disk layout, as a `SupervisorRunReader`. */
|
|
839
|
-
declare function loopsSupervisorRunReader(runDir: string, opts?: LoopsReaderOptions): SupervisorRunReader;
|
|
840
|
-
/**
|
|
841
|
-
* Analyze a supervisor run. Accepts a run directory (read through the loops
|
|
842
|
-
* reader), any `SupervisorRunReader`, or already-read source bytes — so a
|
|
843
|
-
* caller with its own store never has to touch the filesystem layout.
|
|
844
|
-
*/
|
|
845
|
-
declare function analyzeSupervisorRun(input: string | SupervisorRunReader | SupervisorRunSources, opts?: LoopsReaderOptions): Promise<SupervisorRunReport>;
|
|
846
|
-
interface WriteSupervisorRunOptions extends LoopsReaderOptions {
|
|
847
|
-
/** Append the headline block here (the experiment's run log). */
|
|
848
|
-
readonly appendHeadlineTo?: string;
|
|
849
|
-
/** Also console.log the headline (default true). */
|
|
850
|
-
readonly echo?: boolean;
|
|
851
|
-
/**
|
|
852
|
-
* Write `run-report.{json,md}` here instead of into the run dir. Set when
|
|
853
|
-
* reporting over a run directory that must stay READ-ONLY (a live run, an
|
|
854
|
-
* archived generation).
|
|
855
|
-
*/
|
|
856
|
-
readonly reportDir?: string;
|
|
857
|
-
}
|
|
858
|
-
/**
|
|
859
|
-
* Read a completed run, write `run-report.json` + `run-report.md` beside its
|
|
860
|
-
* artifacts, and append the headline block to the run log. Never throws on a
|
|
861
|
-
* missing artifact — a run that produced nothing still yields a report whose
|
|
862
|
-
* every metric says why.
|
|
863
|
-
*/
|
|
864
|
-
declare function writeSupervisorRunReport(runDir: string, opts?: WriteSupervisorRunOptions): Promise<SupervisorRunReport>;
|
|
865
|
-
/**
|
|
866
|
-
* File stem for out-of-tree reports. Built from the run path's identifying
|
|
867
|
-
* segments — candidate tag (the segment under `arm-runs/`), rep, instance, arm
|
|
868
|
-
* — so two runs of the same instance from different candidates/reps never
|
|
869
|
-
* overwrite each other.
|
|
870
|
-
*/
|
|
871
|
-
declare function supervisorReportStem(runDir: string): string;
|
|
872
|
-
/**
|
|
873
|
-
* Best-effort wrapper for a hot path: a reporting failure must never kill a run
|
|
874
|
-
* that already produced real work. Returns null and logs the reason instead.
|
|
875
|
-
*/
|
|
876
|
-
declare function writeSupervisorRunReportSafe(runDir: string, opts?: WriteSupervisorRunOptions): Promise<SupervisorRunReport | null>;
|
|
877
|
-
/**
|
|
878
|
-
* Report every run under an experiment `outDir` (any depth of
|
|
879
|
-
* `runs/<iid>/<arm>`), write each run's report, and write the rollup at
|
|
880
|
-
* `<outDir>/run-report-round.{json,md}`.
|
|
881
|
-
*/
|
|
882
|
-
declare function reportSupervisorRound(outDir: string, opts?: WriteSupervisorRunOptions & {
|
|
883
|
-
title?: string;
|
|
884
|
-
}): Promise<SupervisorRunRollup>;
|
|
885
|
-
/** Every `<...>/runs/<iid>/<arm>` directory under `root`. */
|
|
886
|
-
declare function findSupervisorRunDirs(root: string): Promise<string[]>;
|
|
887
|
-
|
|
888
|
-
/**
|
|
889
|
-
* Human-readable renderings of a supervisor-run report. Zero and unavailable
|
|
890
|
-
* render differently on purpose (`0` vs `unavailable — <reason>`), because the
|
|
891
|
-
* two have driven opposite conclusions about the same architecture.
|
|
892
|
-
*/
|
|
893
|
-
|
|
894
|
-
/**
|
|
895
|
-
* The block appended to a run log after every run — the answers an operator asks
|
|
896
|
-
* for, in the log tail, with no extra command.
|
|
897
|
-
*/
|
|
898
|
-
declare function renderSupervisorRunHeadline(r: SupervisorRunReport): string;
|
|
899
|
-
declare function renderSupervisorRunMarkdown(r: SupervisorRunReport): string;
|
|
900
|
-
declare function renderSupervisorRollupMarkdown(rollup: SupervisorRunRollup, title?: string): string;
|
|
901
|
-
|
|
902
|
-
/**
|
|
903
|
-
* The supervision tree as `tangle.rollout.v1` rows.
|
|
904
|
-
*
|
|
905
|
-
* A supervisor run IS a tree of rollouts, so its nodes are not a new shape:
|
|
906
|
-
* the root becomes one `RolloutLine` with `role: 'supervisor'`, every spawned
|
|
907
|
-
* worker becomes a `RolloutLine` with `role: 'worker'` and
|
|
908
|
-
* `parent_rollout_id` pointing at its spawner. The rows append to the same
|
|
909
|
-
* ledger as solo-agent rollouts and join to them with the same keys.
|
|
910
|
-
*
|
|
911
|
-
* What the journal CANNOT supply is the transcript: a worker's messages live
|
|
912
|
-
* in its harness store (opencode sqlite, Claude Code jsonl), which the
|
|
913
|
-
* `src/rollout/readers/*` intake readers own. Rows minted here are therefore
|
|
914
|
-
* GAP lines (`messages: []`, `provenance.gap` set) carrying identity,
|
|
915
|
-
* structure, outcome and cost; hydrating them with messages is the readers'
|
|
916
|
-
* job, keyed on `artifacts.transcript_ref`.
|
|
917
|
-
*
|
|
918
|
-
* Timing lives in `outcome.metrics` (`spawned_at` / `settled_at` / `wall_ms`)
|
|
919
|
-
* rather than a schema field: `tangle.rollout.v1` describes ONE invocation,
|
|
920
|
-
* and the inter-invocation event timeline — which is what waves, concurrency,
|
|
921
|
-
* idle and utilization are computed from — is a property of the journal, not
|
|
922
|
-
* of any single row. The analyzer reads that timeline; these rows carry the
|
|
923
|
-
* per-node facts.
|
|
924
|
-
*/
|
|
925
|
-
|
|
926
|
-
interface SupervisorRolloutOptions {
|
|
927
|
-
/** Benchmark/suite id for `task.suite`. Defaults to `'supervisor-run'`. */
|
|
928
|
-
readonly suite?: string;
|
|
929
|
-
/** `task.split`. Defaults to `'search'` (the trainable pool). */
|
|
930
|
-
readonly split?: RolloutSplit;
|
|
931
|
-
/** Replicate index. Defaults to 0. */
|
|
932
|
-
readonly rep?: number;
|
|
933
|
-
/** Sampling seed the campaign pinned. Defaults to null (not recorded). */
|
|
934
|
-
readonly seed?: number | null;
|
|
935
|
-
/** `run_id` for every node. Defaults to the supervisor root id, else `runRef`. */
|
|
936
|
-
readonly runId?: string;
|
|
937
|
-
/** Harness that drove the supervisor. */
|
|
938
|
-
readonly supervisorHarness?: string | null;
|
|
939
|
-
/** Harness that drove the workers. */
|
|
940
|
-
readonly workerHarness?: string | null;
|
|
941
|
-
/** Model the supervisor ran on. */
|
|
942
|
-
readonly supervisorModel?: string | null;
|
|
943
|
-
/** Model the workers ran on. */
|
|
944
|
-
readonly workerModel?: string | null;
|
|
945
|
-
readonly experimentId?: string | null;
|
|
946
|
-
readonly candidateId?: string | null;
|
|
947
|
-
readonly generation?: number | null;
|
|
948
|
-
readonly candidateIndex?: number | null;
|
|
949
|
-
/** Pins `provenance.captured_at`; defaults to now. */
|
|
950
|
-
readonly capturedAt?: string;
|
|
951
|
-
}
|
|
952
|
-
/**
|
|
953
|
-
* Mint the supervision tree as rollout rows. Returns the rows plus the gaps
|
|
954
|
-
* that made any of them incomplete — same unavailable-vs-zero discipline as
|
|
955
|
-
* the report: a row with no transcript says WHY, it never pretends to be empty.
|
|
956
|
-
*/
|
|
957
|
-
declare function supervisorRunRolloutLines(src: SupervisorRunSources, opts?: SupervisorRolloutOptions): SupervisorRunTree;
|
|
958
|
-
|
|
959
|
-
export { type ClaudeCodeReaderOptions, type CloseRow, DEFAULT_CANCEL_TOOLS, DEFAULT_SPAWN_TOOLS, DEFAULT_STEER_TOOLS, type DecisionMetrics, type EconomicsMetrics, type LoopsReaderOptions, type Measured, NO_SOURCE_LIMITS, type OrchestrationMetrics, type OutcomeMetrics, type PatchStats, type PerWorkerRow, type RoleSpend, type RollupCellRow, SUPERVISOR_RUN_ROLLUP_SCHEMA, SUPERVISOR_RUN_SCHEMA, type SourceLimits, type SpawnRow, type SteerBreakdown, type SupervisorRolloutOptions, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSources, type SupervisorRunTree, type SupervisorTreeFacts, type Unavailable, type WallDistribution, type WorkerLogFacts, type WorkerLogSource, type WriteSupervisorRunOptions, analyzeSupervisorRun, analyzeSupervisorRunSources, claudeCodeSupervisorRunReader, findSupervisorRunDirIn, findSupervisorRunDirs, isUnavailable, loopsSupervisorRunReader, parsePatch, parseSupervisorTree, readClaudeCodeSupervisorRun, readLoopsSupervisorRun, renderSupervisorRollupMarkdown, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, reportSupervisorRound, rollupSupervisorRuns, showMeasured, supervisorReportStem, supervisorRunRolloutLines, unavailable, writeSupervisorRunReport, writeSupervisorRunReportSafe };
|
|
1
|
+
import { $ as isUnavailable, A as rollupSupervisorRuns, B as RollupCellRow, C as CloseRow, D as analyzeSupervisorRunSources, E as WorkerLogFacts, F as OrchestrationMetrics, G as SupervisorRunReader, H as SUPERVISOR_RUN_SCHEMA, I as OutcomeMetrics, J as SupervisorRunSources, K as SupervisorRunReport, L as PatchStats, M as EconomicsMetrics, N as Measured, O as parsePatch, P as NO_SOURCE_LIMITS, Q as WorkerLogSource, R as PerWorkerRow, S as readClaudeCodeSupervisorRun, T as SupervisorTreeFacts, U as SourceLimits, V as SUPERVISOR_RUN_ROLLUP_SCHEMA, W as SteerBreakdown, X as Unavailable, Y as SupervisorRunTree, Z as WallDistribution, _ as ClaudeCodeReaderOptions, a as renderSupervisorRunMarkdown, b as DEFAULT_STEER_TOOLS, c as analyzeSupervisorRun, d as loopsSupervisorRunReader, et as showMeasured, f as readLoopsSupervisorRun, g as writeSupervisorRunReportSafe, h as writeSupervisorRunReport, i as renderSupervisorRunHeadline, j as DecisionMetrics, k as parseSupervisorTree, l as findSupervisorRunDirIn, m as supervisorReportStem, n as supervisorRunRolloutLines, o as LoopsReaderOptions, p as reportSupervisorRound, q as SupervisorRunRollup, r as renderSupervisorRollupMarkdown, s as WriteSupervisorRunOptions, t as SupervisorRolloutOptions, tt as unavailable, u as findSupervisorRunDirs, v as DEFAULT_CANCEL_TOOLS, w as SpawnRow, x as claudeCodeSupervisorRunReader, y as DEFAULT_SPAWN_TOOLS, z as RoleSpend } from "../index-C61Wi7yg.js";
|
|
2
|
+
export { type ClaudeCodeReaderOptions, type CloseRow, DEFAULT_CANCEL_TOOLS, DEFAULT_SPAWN_TOOLS, DEFAULT_STEER_TOOLS, type DecisionMetrics, type EconomicsMetrics, type LoopsReaderOptions, type Measured, NO_SOURCE_LIMITS, type OrchestrationMetrics, type OutcomeMetrics, type PatchStats, type PerWorkerRow, type RoleSpend, type RollupCellRow, SUPERVISOR_RUN_ROLLUP_SCHEMA, SUPERVISOR_RUN_SCHEMA, type SourceLimits, type SpawnRow, type SteerBreakdown, type SupervisorRolloutOptions, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSources, type SupervisorRunTree, type SupervisorTreeFacts, type Unavailable, type WallDistribution, type WorkerLogFacts, type WorkerLogSource, type WriteSupervisorRunOptions, analyzeSupervisorRun, analyzeSupervisorRunSources, claudeCodeSupervisorRunReader, findSupervisorRunDirIn, findSupervisorRunDirs, isUnavailable, loopsSupervisorRunReader, parsePatch, parseSupervisorTree, readClaudeCodeSupervisorRun, readLoopsSupervisorRun, renderSupervisorRollupMarkdown, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, reportSupervisorRound, rollupSupervisorRuns, showMeasured, supervisorReportStem, supervisorRunRolloutLines, unavailable, writeSupervisorRunReport, writeSupervisorRunReportSafe };
|