@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
@@ -1,1379 +1,622 @@
1
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
2
- interface BudgetSpec {
3
- tokens?: number;
4
- wallMs?: number;
5
- calls?: number;
6
- usd?: number;
7
- }
8
- interface RunOutcome$1 {
9
- score?: number;
10
- pass?: boolean;
11
- failureClass?: FailureClass;
12
- notes?: string;
13
- }
14
- /**
15
- * Layer — optional classification in a nested build workflow.
16
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
17
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
18
- * `app-runtime`: a run of the generated agent against a domain scenario.
19
- * `meta`: any meta-eval (judge replay, correlation analysis).
20
- */
21
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
22
- interface Run {
23
- runId: string;
24
- /**
25
- * Stable identifier of the scenario being executed.
26
- *
27
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
28
- * input WITHOUT this field, substituting a sensible default
29
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
30
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
31
- * keeps the persisted shape unambiguous for downstream filters + aggregations
32
- * while removing the boilerplate of inventing placeholder ids at the call site.
33
- */
34
- scenarioId: string;
35
- variantId?: string;
36
- datasetVersion?: string;
37
- /** Git SHA of agent code at run time. */
38
- codeSha?: string;
39
- /** Hash of the prompt template + any system prompt. */
40
- promptSha?: string;
41
- /** Model id + date + system-prompt hash, concatenated. */
42
- modelFingerprint?: string;
43
- seed?: number;
44
- /** Arbitrary environment markers (shell, docker version, tz). */
45
- envFingerprint?: Record<string, string>;
46
- /** Version of the redaction rules applied to this run. */
47
- redactionVersion?: string;
48
- /** Parent run in a nested build workflow. A builder run's children are
49
- * app-build runs; those children are app-runtime runs. */
50
- parentRunId?: string;
51
- /** Stable project identifier — groups runs across chats + sessions. */
52
- projectId?: string;
53
- /** Chat/conversation identifier within a project. */
54
- chatId?: string;
55
- /** Layer classification — hint for aggregation; not enforced. */
56
- layer?: RunLayer;
57
- startedAt: number;
58
- endedAt?: number;
59
- status: RunStatus;
60
- outcome?: RunOutcome$1;
61
- budget?: BudgetSpec;
62
- /** Free-form labels for downstream grouping. */
63
- tags?: Record<string, string>;
64
- }
65
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
66
- type SpanStatus = 'ok' | 'error';
67
- interface SpanBase {
68
- spanId: string;
69
- parentSpanId?: string;
70
- runId: string;
71
- kind: SpanKind;
72
- name: string;
73
- startedAt: number;
74
- endedAt?: number;
75
- status?: SpanStatus;
76
- error?: string;
77
- /** Anything not covered by typed fields. Kept deliberately free-form. */
78
- attributes?: Record<string, unknown>;
79
- }
80
- interface Message {
81
- role: 'system' | 'user' | 'assistant' | 'tool';
82
- content: string;
83
- tokens?: number;
84
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
85
- images?: Array<{
86
- artifactId?: string;
87
- url?: string;
88
- mime?: string;
89
- }>;
90
- }
91
- interface LlmSpan extends SpanBase {
92
- kind: 'llm';
93
- model: string;
94
- messages: Message[];
95
- output?: string;
96
- inputTokens?: number;
97
- /** All generated tokens, including the reasoning subset when present. */
98
- outputTokens?: number;
99
- cachedTokens?: number;
100
- cacheWriteTokens?: number;
101
- /** Reasoning-token subset of `outputTokens`. */
102
- reasoningTokens?: number;
103
- costUsd?: number;
104
- finishReason?: string;
105
- }
106
- interface ToolSpan extends SpanBase {
107
- kind: 'tool';
108
- toolName: string;
109
- args: unknown;
110
- /** False when the source observed the call but did not capture its arguments. */
111
- argsCaptured?: boolean;
112
- result?: unknown;
113
- latencyMs?: number;
114
- }
115
- interface RetrievalSpan extends SpanBase {
116
- kind: 'retrieval';
117
- query: string;
118
- hits: Array<{
119
- docId: string;
120
- score: number;
121
- content?: string;
122
- }>;
123
- }
124
- interface JudgeSpan extends SpanBase {
125
- kind: 'judge';
126
- judgeId: string;
127
- /** Span this judgment applies to. */
128
- targetSpanId: string;
129
- dimension: string;
130
- /** Numeric score (free-range; interpretation up to the judge). */
131
- score: number;
132
- rationale?: string;
133
- evidence?: string;
134
- }
135
- interface SandboxSpan extends SpanBase {
136
- kind: 'sandbox';
137
- image?: string;
138
- command?: string;
139
- exitCode?: number;
140
- testsTotal?: number;
141
- testsPassed?: number;
142
- stdoutHash?: string;
143
- stderrHash?: string;
144
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
145
- wallMs?: number;
146
- }
147
- interface GenericSpan extends SpanBase {
148
- kind: 'agent' | 'custom';
149
- }
150
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
151
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
152
- interface TraceEvent {
153
- eventId: string;
154
- runId: string;
155
- spanId?: string;
156
- kind: EventKind;
157
- timestamp: number;
158
- payload: Record<string, unknown>;
159
- }
160
- interface BudgetLedgerEntry {
161
- runId: string;
162
- dimension: keyof BudgetSpec;
163
- limit: number;
164
- consumed: number;
165
- remaining: number;
166
- timestamp: number;
167
- breached: boolean;
168
- /** Span that triggered this entry, if any. */
169
- spanId?: string;
170
- }
171
- interface Artifact {
172
- artifactId: string;
173
- runId: string;
174
- spanId?: string;
175
- contentType: string;
176
- sizeBytes: number;
177
- /** sha256 in hex. */
178
- hash: string;
179
- /** External storage URL (R2, S3, filesystem path). */
180
- storageUrl?: string;
181
- /** Inline content for small blobs — keep under ~64KB. */
182
- inlineContent?: string;
183
- }
184
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
185
-
186
- interface RunFilter {
187
- scenarioId?: string;
188
- variantId?: string;
189
- status?: RunStatus;
190
- since?: number;
191
- until?: number;
192
- tag?: {
193
- key: string;
194
- value: string;
195
- };
196
- parentRunId?: string;
197
- projectId?: string;
198
- chatId?: string;
199
- layer?: RunLayer;
200
- }
201
- interface SpanFilter {
202
- runId?: string;
203
- parentSpanId?: string;
204
- kind?: SpanKind;
205
- name?: string;
206
- toolName?: string;
207
- judgeId?: string;
208
- since?: number;
209
- until?: number;
210
- }
211
- interface EventFilter {
212
- runId?: string;
213
- spanId?: string;
214
- kind?: EventKind;
215
- since?: number;
216
- until?: number;
217
- }
218
- interface TraceStore {
219
- appendRun(run: Run): Promise<void>;
220
- updateRun(runId: string, patch: Partial<Run>): Promise<void>;
221
- appendSpan(span: Span): Promise<void>;
222
- updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
223
- appendEvent(event: TraceEvent): Promise<void>;
224
- appendArtifact(artifact: Artifact): Promise<void>;
225
- appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
226
- getRun(runId: string): Promise<Run | undefined>;
227
- listRuns(filter?: RunFilter): Promise<Run[]>;
228
- spans(filter?: SpanFilter): Promise<Span[]>;
229
- events(filter?: EventFilter): Promise<TraceEvent[]>;
230
- budget(runId: string): Promise<BudgetLedgerEntry[]>;
231
- artifacts(runId: string): Promise<Artifact[]>;
232
- }
233
-
234
- /**
235
- * Calibration curve — binned "if eval says X, what does reality show?"
236
- *
237
- * Companion to correlationStudy. Raw correlation is a single number;
238
- * the calibration curve shows *where* the eval is well-calibrated vs
239
- * overconfident / underconfident. Buckets the eval metric, computes
240
- * mean outcome per bucket, reports expected-calibration-error (ECE).
241
- */
242
-
243
- interface CalibrationBin {
244
- lower: number;
245
- upper: number;
246
- n: number;
247
- evalMean: number;
248
- outcomeMean: number;
249
- /** |outcomeMean − evalMean|; contributes to ECE weighted by n/total. */
250
- gap: number;
251
- }
252
- interface CalibrationReport {
253
- evalMetric: string;
254
- outcomeMetric: string;
255
- n: number;
256
- bins: CalibrationBin[];
257
- /** Expected Calibration Error — Σ (n_i/N) × |outcomeMean_i − evalMean_i|. */
258
- ece: number;
259
- /** Max bin gap — upper bound on miscalibration. */
260
- maxGap: number;
261
- }
262
-
263
- /**
264
- * Off-policy evaluation primitives.
265
- *
266
- * Standard inverse-probability-weighted (IPS), self-normalized
267
- * importance-weighted (SNIPS), and doubly-robust (DR) estimators for the
268
- * value of a *target* policy given trajectories collected under a
269
- * *behavior* policy. This is the canonical RL eval task: "we have last
270
- * week's runs, we changed the policy — how would the new one do without
271
- * re-running?"
272
- *
273
- * The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan
274
- * & Joachims 2015 for SNIPS) but the *application* to LLM-agent
275
- * evaluation needs care:
276
- *
277
- * - The "policy" is the (prompt, tool config, model snapshot) triple.
278
- * Two policies have the same probability over an action *iff* their
279
- * LLM call would emit the same token with the same probability —
280
- * which is generally unknowable without the model log-probs.
281
- * - For LLM agents, propensity scores must be supplied by the caller
282
- * (logged in the trace, recovered from token log-probs, or estimated
283
- * via a learned propensity model). We do NOT estimate propensity here.
284
- * - Doubly-robust requires two outputs from a Q-function: its prediction
285
- * for the logged action and its expectation under the target policy.
286
- * Consumers compute these with a tabular estimate, regression fit, or
287
- * learned reward model before constructing the trajectories.
288
- *
289
- * Bias / variance tradeoffs:
290
- * - IPS: unbiased; high variance for small overlap, infinite variance
291
- * when target has support outside behavior.
292
- * - SNIPS: lower variance, slight bias; usually preferred in practice.
293
- * - DR: doubly-robust — unbiased if either propensity OR Q-function is
294
- * correct. Lowest practical variance when Q is decent. Use this.
295
- *
296
- * Caveat the panel will land: on the LLM-agent setting, propensity scores
297
- * recovered from token log-probs are noisy, the action space is enormous,
298
- * and overlap is often poor. These estimators are useful but not magic;
299
- * complement with `replayCampaign` (exact replay where the request hashes
300
- * match) for high-confidence answers and OPE for the gap.
301
- */
302
- interface OffPolicyTrajectory {
303
- /** Stable id, for traceability through the dataset. */
304
- runId: string;
305
- /** Reward observed under the behavior policy (the realized outcome). */
306
- reward: number;
307
- /**
308
- * Behavior-policy probability of the action that was taken. For LLM
309
- * agents this is typically `exp(sum(token_log_probs))` over the chosen
310
- * trajectory. Must be in (0, 1].
311
- */
312
- behaviorProb: number;
313
- /**
314
- * Target-policy probability of the same action. For replay-style
315
- * counterfactual evaluation this is what the *new* policy would have
316
- * assigned to the *old* trajectory. Must be in [0, 1].
317
- */
318
- targetProb: number;
319
- /**
320
- * Model-based reward prediction for the action selected by the behavior
321
- * policy: `Q_hat(context, loggedAction)`. Supply this together with
322
- * `vHatTarget` for contextual-bandit doubly-robust estimation.
323
- */
324
- qHatChosen?: number | null;
325
- /**
326
- * Expected model-based reward under the target policy:
327
- * `sum_action targetPolicy(action | context) * Q_hat(context, action)`.
328
- * Supply this together with `qHatChosen`. For an honest evaluation, both
329
- * values must come from a model cross-fitted or trained outside this row.
330
- */
331
- vHatTarget?: number | null;
332
- /**
333
- * @deprecated Use `qHatChosen` and `vHatTarget` together. When the new pair
334
- * is absent, this scalar is used as both terms to preserve existing results.
335
- * When the new pair is present, this field is ignored.
336
- */
337
- qHat?: number | null;
338
- }
339
- interface OffPolicyContributionCounts {
340
- /** Contributions using the contextual-bandit doubly-robust formula. */
341
- dr: number;
342
- /** Contributions using exact IPS because no reward-model estimate was supplied. */
343
- ipsFallback: number;
344
- /** Contributions using the deprecated single-scalar formula. */
345
- legacyScalar: number;
346
- }
347
- interface OffPolicyEstimate {
348
- /** Estimated value of the target policy. */
349
- value: number;
350
- /** Standard error of the estimate. */
351
- standardError: number;
352
- /** Effective sample size (Kong 1992). Lower = more reliance on a few high-weight samples. */
353
- effectiveSampleSize: number;
354
- /** Number of trajectories used. */
355
- n: number;
356
- /**
357
- * Diagnostic: maximum importance weight observed. Large values (>>10x
358
- * mean) are a red flag — variance is dominated by a few outliers.
359
- */
360
- maxImportanceWeight: number;
361
- /** Populated by `doublyRobust` to expose which formula each row used. */
362
- contributionCounts?: OffPolicyContributionCounts;
363
- }
364
- interface OffPolicyOptions {
365
- /**
366
- * Cap importance weights at this value (Ionides 2008 truncated IS) to
367
- * trade unbiasedness for variance reduction. Default `Infinity` (no cap).
368
- * Set e.g. `10` for stable estimates when the policies are close.
369
- */
370
- weightCap?: number;
371
- /** Reward clipping range. Default `[0, 1]`. */
372
- rewardClip?: {
373
- low: number;
374
- high: number;
375
- };
376
- }
377
-
378
- declare const BELIEF_DECISION_KINDS: readonly ["continue", "verify", "ask", "retry", "stop", "memory-write", "memory-read", "tool-select", "skill-select", "workflow-select", "surface-promote"];
1
+ import { a as RunRecord, s as RunSplitTag } from "../run-record-CnZu_gjl.js";
2
+ import { s as TraceStore } from "../store-CT9YIIve.js";
3
+ import { w as CalibrationReport } from "../index-6N0aYmpW.js";
4
+ import { i as OffPolicyTrajectory, n as OffPolicyEstimate, r as OffPolicyOptions } from "../off-policy-mskQw8Mb.js";
5
+ import { i as CodeAgentSessionMetrics, n as CodeAgentSessionIntakeOptions, t as CodeAgentSessionDiagnostic, v as CodeAgentSessionObservation, y as CodeAgentSessionSource } from "../code-agent-session-DqqgOJaz.js";
6
+ import { a as RuntimeTrajectoryRecord, n as RuntimeTrajectoryEvidenceProjection, t as ProjectRuntimeTrajectoryEvidenceOptions } from "../runtime-trajectory-BvSZcCHD.js";
7
+ //#region src/belief-state/types.d.ts
8
+ declare const BELIEF_DECISION_KINDS: readonly ['continue', 'verify', 'ask', 'retry', 'stop', 'memory-write', 'memory-read', 'tool-select', 'skill-select', 'workflow-select', 'surface-promote'];
379
9
  type BeliefDecisionKind = (typeof BELIEF_DECISION_KINDS)[number];
380
- declare const BELIEF_EVIDENCE_SOURCES: readonly ["run", "span", "event", "finding", "memory", "knowledge", "policy"];
10
+ declare const BELIEF_EVIDENCE_SOURCES: readonly ['run', 'span', 'event', 'finding', 'memory', 'knowledge', 'policy'];
381
11
  type BeliefEvidenceSource = (typeof BELIEF_EVIDENCE_SOURCES)[number];
382
- declare const BELIEF_EVIDENCE_QUALITIES: readonly ["direct", "derived", "self-reported", "unverified", "stale", "contradicted"];
12
+ declare const BELIEF_EVIDENCE_QUALITIES: readonly ['direct', 'derived', 'self-reported', 'unverified', 'stale', 'contradicted'];
383
13
  type BeliefEvidenceQuality = (typeof BELIEF_EVIDENCE_QUALITIES)[number];
384
14
  declare const BELIEF_EVALUATION_CRITERIA: readonly [{
385
- readonly id: "capture-integrity";
386
- readonly label: "Capture integrity";
387
- readonly reasonCodes: readonly ["trace-missing", "run-record-missing", "backend-integrity-missing"];
15
+ readonly id: 'capture-integrity';
16
+ readonly label: 'Capture integrity';
17
+ readonly reasonCodes: readonly ['trace-missing', 'run-record-missing', 'backend-integrity-missing'];
388
18
  }, {
389
- readonly id: "decision-completeness";
390
- readonly label: "Decision completeness";
391
- readonly reasonCodes: readonly ["candidate-actions-missing", "chosen-action-missing", "decision-evidence-missing"];
19
+ readonly id: 'decision-completeness';
20
+ readonly label: 'Decision completeness';
21
+ readonly reasonCodes: readonly ['candidate-actions-missing', 'chosen-action-missing', 'decision-evidence-missing'];
392
22
  }, {
393
- readonly id: "evidence-quality";
394
- readonly label: "Evidence quality";
395
- readonly reasonCodes: readonly ["evidence-stale", "evidence-contradictory", "evidence-unverified", "evidence-self-reported"];
23
+ readonly id: 'evidence-quality';
24
+ readonly label: 'Evidence quality';
25
+ readonly reasonCodes: readonly ['evidence-stale', 'evidence-contradictory', 'evidence-unverified', 'evidence-self-reported'];
396
26
  }, {
397
- readonly id: "outcome-quality";
398
- readonly label: "Outcome quality";
399
- readonly reasonCodes: readonly ["outcome-missing", "outcome-delayed", "cost-missing"];
27
+ readonly id: 'outcome-quality';
28
+ readonly label: 'Outcome quality';
29
+ readonly reasonCodes: readonly ['outcome-missing', 'outcome-delayed', 'cost-missing'];
400
30
  }, {
401
- readonly id: "calibration";
402
- readonly label: "Calibration";
403
- readonly reasonCodes: readonly ["confidence-missing", "calibration-unsupported", "calibration-gap-high"];
31
+ readonly id: 'calibration';
32
+ readonly label: 'Calibration';
33
+ readonly reasonCodes: readonly ['confidence-missing', 'calibration-unsupported', 'calibration-gap-high'];
404
34
  }, {
405
- readonly id: "accepted-region-risk";
406
- readonly label: "Accepted-region risk";
407
- readonly reasonCodes: readonly ["accepted-error-high", "coverage-too-low"];
35
+ readonly id: 'accepted-region-risk';
36
+ readonly label: 'Accepted-region risk';
37
+ readonly reasonCodes: readonly ['accepted-error-high', 'coverage-too-low'];
408
38
  }, {
409
- readonly id: "policy-value";
410
- readonly label: "Policy value";
411
- readonly reasonCodes: readonly ["utility-lift-missing", "baseline-dominates", "cost-too-high"];
39
+ readonly id: 'policy-value';
40
+ readonly label: 'Policy value';
41
+ readonly reasonCodes: readonly ['utility-lift-missing', 'baseline-dominates', 'cost-too-high'];
412
42
  }, {
413
- readonly id: "ope-support";
414
- readonly label: "OPE support";
415
- readonly reasonCodes: readonly ["behavior-propensity-missing", "behavior-propensity-invalid", "target-propensity-missing", "target-propensity-invalid", "effective-sample-size-low", "importance-weight-high"];
43
+ readonly id: 'ope-support';
44
+ readonly label: 'OPE support';
45
+ readonly reasonCodes: readonly ['behavior-propensity-missing', 'behavior-propensity-invalid', 'target-propensity-missing', 'target-propensity-invalid', 'effective-sample-size-low', 'importance-weight-high'];
416
46
  }, {
417
- readonly id: "memory-health";
418
- readonly label: "Memory health";
419
- readonly reasonCodes: readonly ["memory-stale", "memory-poisoning-risk", "context-bloat", "memory-write-unverified"];
47
+ readonly id: 'memory-health';
48
+ readonly label: 'Memory health';
49
+ readonly reasonCodes: readonly ['memory-stale', 'memory-poisoning-risk', 'context-bloat', 'memory-write-unverified'];
420
50
  }, {
421
- readonly id: "surface-attribution";
422
- readonly label: "Surface attribution";
423
- readonly reasonCodes: readonly ["surface-claim-unsupported", "causal-attribution-missing"];
51
+ readonly id: 'surface-attribution';
52
+ readonly label: 'Surface attribution';
53
+ readonly reasonCodes: readonly ['surface-claim-unsupported', 'causal-attribution-missing'];
424
54
  }, {
425
- readonly id: "generalization";
426
- readonly label: "Generalization";
427
- readonly reasonCodes: readonly ["split-missing", "holdout-regression", "task-family-coverage-low", "leakage-risk"];
55
+ readonly id: 'generalization';
56
+ readonly label: 'Generalization';
57
+ readonly reasonCodes: readonly ['split-missing', 'holdout-regression', 'task-family-coverage-low', 'leakage-risk'];
428
58
  }, {
429
- readonly id: "promotion";
430
- readonly label: "Promotion";
431
- readonly reasonCodes: readonly ["negative-control-failed", "promotion-gate-failed", "human-review-required"];
59
+ readonly id: 'promotion';
60
+ readonly label: 'Promotion';
61
+ readonly reasonCodes: readonly ['negative-control-failed', 'promotion-gate-failed', 'human-review-required'];
432
62
  }];
433
63
  type BeliefEvaluationCriterionId = (typeof BELIEF_EVALUATION_CRITERIA)[number]['id'];
434
64
  type BeliefDecisionReasonCode = (typeof BELIEF_EVALUATION_CRITERIA)[number]['reasonCodes'][number];
435
65
  interface BeliefDecisionReason {
436
- code: BeliefDecisionReasonCode;
437
- criterion?: BeliefEvaluationCriterionId;
438
- detail?: string;
439
- evidenceIds?: string[];
440
- metadata?: Record<string, unknown>;
66
+ code: BeliefDecisionReasonCode;
67
+ criterion?: BeliefEvaluationCriterionId;
68
+ detail?: string;
69
+ evidenceIds?: string[];
70
+ metadata?: Record<string, unknown>;
441
71
  }
442
72
  declare function isBeliefDecisionKind(value: unknown): value is BeliefDecisionKind;
443
73
  declare function isBeliefEvidenceSource(value: unknown): value is BeliefEvidenceSource;
444
74
  interface BeliefEvidenceRef {
445
- source: BeliefEvidenceSource;
446
- id: string;
447
- runId?: string;
448
- spanId?: string;
449
- eventId?: string;
450
- detail?: string;
451
- quality?: BeliefEvidenceQuality;
452
- observedAt?: string;
453
- metadata?: Record<string, unknown>;
75
+ source: BeliefEvidenceSource;
76
+ id: string;
77
+ runId?: string;
78
+ spanId?: string;
79
+ eventId?: string;
80
+ detail?: string;
81
+ quality?: BeliefEvidenceQuality;
82
+ observedAt?: string;
83
+ metadata?: Record<string, unknown>;
454
84
  }
455
85
  interface BeliefDecisionOutcome {
456
- success?: boolean;
457
- score?: number;
458
- reward?: number;
459
- costUsd?: number;
460
- observedAt?: string;
461
- metadata?: Record<string, unknown>;
86
+ success?: boolean;
87
+ score?: number;
88
+ reward?: number;
89
+ costUsd?: number;
90
+ observedAt?: string;
91
+ metadata?: Record<string, unknown>;
462
92
  }
463
93
  interface BeliefDecisionPoint {
464
- id: string;
465
- runId: string;
466
- scenarioId?: string;
467
- stepIndex: number;
468
- kind: BeliefDecisionKind;
469
- chosenAction: string;
470
- candidateActions?: string[];
471
- confidence?: number;
472
- behaviorProb?: number;
473
- targetProb?: number;
474
- qHatChosen?: number | null;
475
- vHatTarget?: number | null;
476
- /** @deprecated Use `qHatChosen` and `vHatTarget` together. */
477
- qHat?: number | null;
478
- costUsd?: number;
479
- evidence: BeliefEvidenceRef[];
480
- outcome?: BeliefDecisionOutcome;
481
- reasons?: BeliefDecisionReason[];
482
- metadata?: Record<string, unknown>;
94
+ id: string;
95
+ runId: string;
96
+ scenarioId?: string;
97
+ stepIndex: number;
98
+ kind: BeliefDecisionKind;
99
+ chosenAction: string;
100
+ candidateActions?: string[];
101
+ confidence?: number;
102
+ behaviorProb?: number;
103
+ targetProb?: number;
104
+ qHatChosen?: number | null;
105
+ vHatTarget?: number | null;
106
+ costUsd?: number;
107
+ evidence: BeliefEvidenceRef[];
108
+ outcome?: BeliefDecisionOutcome;
109
+ reasons?: BeliefDecisionReason[];
110
+ metadata?: Record<string, unknown>;
483
111
  }
484
112
  interface BeliefDecisionExtractionDiagnostic {
485
- runId: string;
486
- eventId?: string;
487
- severity: 'info' | 'warning' | 'error';
488
- reason: string;
113
+ runId: string;
114
+ eventId?: string;
115
+ severity: 'info' | 'warning' | 'error';
116
+ reason: string;
489
117
  }
490
118
  interface BeliefDecisionExtractionReport {
491
- decisions: BeliefDecisionPoint[];
492
- diagnostics: BeliefDecisionExtractionDiagnostic[];
119
+ decisions: BeliefDecisionPoint[];
120
+ diagnostics: BeliefDecisionExtractionDiagnostic[];
493
121
  }
494
122
  type BeliefPolicyAction = 'accept' | 'defer' | 'verify' | 'ask' | 'retry' | 'stop';
495
123
  interface BeliefPolicyDecision {
496
- action: BeliefPolicyAction;
497
- confidence?: number;
498
- targetProb?: number;
499
- qHatChosen?: number | null;
500
- vHatTarget?: number | null;
501
- /** @deprecated Use `qHatChosen` and `vHatTarget` together. */
502
- qHat?: number | null;
503
- reason?: string;
504
- reasons?: BeliefDecisionReason[];
124
+ action: BeliefPolicyAction;
125
+ confidence?: number;
126
+ targetProb?: number;
127
+ qHatChosen?: number | null;
128
+ vHatTarget?: number | null;
129
+ reason?: string;
130
+ reasons?: BeliefDecisionReason[];
505
131
  }
506
132
  interface BeliefSelectivePolicy {
507
- id: string;
508
- decide(point: BeliefDecisionPoint): BeliefPolicyDecision;
133
+ id: string;
134
+ decide(point: BeliefDecisionPoint): BeliefPolicyDecision;
509
135
  }
510
136
  interface BeliefOpeTargetPolicy {
511
- id: string;
512
- targetProbOf(point: BeliefDecisionPoint): number | null | undefined;
513
- qHatChosenOf?(point: BeliefDecisionPoint): number | null | undefined;
514
- vHatTargetOf?(point: BeliefDecisionPoint): number | null | undefined;
515
- /** @deprecated Use `qHatChosenOf` and `vHatTargetOf` together. */
516
- qHatOf?(point: BeliefDecisionPoint): number | null | undefined;
137
+ id: string;
138
+ targetProbOf(point: BeliefDecisionPoint): number | null | undefined;
139
+ qHatChosenOf?(point: BeliefDecisionPoint): number | null | undefined;
140
+ vHatTargetOf?(point: BeliefDecisionPoint): number | null | undefined;
517
141
  }
518
142
  interface BeliefUtilityOptions {
519
- successUtility?: number;
520
- failureUtility?: number;
521
- deferUtility?: number;
522
- verifyCost?: number;
523
- askCost?: number;
524
- retryCost?: number;
525
- stopUtility?: number;
526
- costWeight?: number;
143
+ successUtility?: number;
144
+ failureUtility?: number;
145
+ deferUtility?: number;
146
+ verifyCost?: number;
147
+ askCost?: number;
148
+ retryCost?: number;
149
+ stopUtility?: number;
150
+ costWeight?: number;
527
151
  }
528
152
  interface BeliefSelectivePolicyMetrics {
529
- policyId: string;
530
- n: number;
531
- accepted: number;
532
- rejected: number;
533
- coverage: number;
534
- acceptedErrorRate: number;
535
- baselineUtility: number;
536
- policyUtility: number;
537
- utilityDelta: number;
538
- utilityCi95: {
539
- mean: number;
540
- lower: number;
541
- upper: number;
542
- };
543
- rejectedMeanReward: number | null;
544
- recommendation: 'ship' | 'hold' | 'need_more_data';
545
- reasons: string[];
153
+ policyId: string;
154
+ n: number;
155
+ accepted: number;
156
+ rejected: number;
157
+ coverage: number;
158
+ acceptedErrorRate: number;
159
+ baselineUtility: number;
160
+ policyUtility: number;
161
+ utilityDelta: number;
162
+ utilityCi95: {
163
+ mean: number;
164
+ lower: number;
165
+ upper: number;
166
+ };
167
+ rejectedMeanReward: number | null;
168
+ recommendation: 'ship' | 'hold' | 'need_more_data';
169
+ reasons: string[];
546
170
  }
547
171
  interface BeliefOpeSupportDiagnostics {
548
- supported: boolean;
549
- n: number;
550
- dropped: number;
551
- effectiveSampleSize: number;
552
- effectiveSampleRatio: number;
553
- maxImportanceWeight: number;
554
- reasons: string[];
172
+ supported: boolean;
173
+ n: number;
174
+ dropped: number;
175
+ effectiveSampleSize: number;
176
+ effectiveSampleRatio: number;
177
+ maxImportanceWeight: number;
178
+ reasons: string[];
555
179
  }
556
180
  interface BeliefOpeReport {
557
- targetPolicyId: string;
558
- ips: OffPolicyEstimate;
559
- snips: OffPolicyEstimate;
560
- dr: OffPolicyEstimate;
561
- support: BeliefOpeSupportDiagnostics;
181
+ targetPolicyId: string;
182
+ ips: OffPolicyEstimate;
183
+ snips: OffPolicyEstimate;
184
+ dr: OffPolicyEstimate;
185
+ support: BeliefOpeSupportDiagnostics;
562
186
  }
563
187
  type BeliefEvaluationStatus = 'ship' | 'hold' | 'need_more_data';
564
188
  type BeliefCalibrationStatus = 'supported' | 'unsupported';
565
189
  type BeliefOpeStatus = 'supported' | 'unsupported' | 'not_requested';
566
190
  interface BeliefPolicyEvaluationReport {
567
- policyId: string;
568
- n: number;
569
- status: BeliefEvaluationStatus;
570
- selectiveStatus: BeliefEvaluationStatus;
571
- calibrationStatus: BeliefCalibrationStatus;
572
- opeStatus: BeliefOpeStatus;
573
- opeTargetPolicyId?: string;
574
- selective: BeliefSelectivePolicyMetrics;
575
- calibration?: CalibrationReport;
576
- ope?: BeliefOpeReport;
577
- diagnostics: string[];
578
- }
579
-
191
+ policyId: string;
192
+ n: number;
193
+ status: BeliefEvaluationStatus;
194
+ selectiveStatus: BeliefEvaluationStatus;
195
+ calibrationStatus: BeliefCalibrationStatus;
196
+ opeStatus: BeliefOpeStatus;
197
+ opeTargetPolicyId?: string;
198
+ selective: BeliefSelectivePolicyMetrics;
199
+ calibration?: CalibrationReport;
200
+ ope?: BeliefOpeReport;
201
+ diagnostics: string[];
202
+ }
203
+ //#endregion
204
+ //#region src/belief-state/calibration.d.ts
580
205
  type BeliefCalibrationRegion = 'all' | 'accepted' | 'rejected';
581
206
  interface BeliefCalibrationOptions {
582
- bins?: number;
583
- minPairs?: number;
584
- policy?: BeliefSelectivePolicy;
585
- region?: BeliefCalibrationRegion;
207
+ bins?: number;
208
+ minPairs?: number;
209
+ policy?: BeliefSelectivePolicy;
210
+ region?: BeliefCalibrationRegion;
586
211
  }
587
212
  declare function calibrateBeliefDecisions(points: BeliefDecisionPoint[], options?: BeliefCalibrationOptions): CalibrationReport | null;
588
-
589
- type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
590
- type AgentProfileDimensionValue = string | number | boolean | null;
591
- interface AgentProfileSource {
592
- /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
593
- kind: string;
594
- /** sha256 over the canonical source profile object. */
595
- hash: string;
596
- }
597
- interface AgentProfileHarness {
598
- id: string;
599
- version?: string;
600
- hash?: string;
601
- }
602
- interface AgentProfileCell {
603
- schemaVersion: AgentProfileCellSchemaVersion;
604
- cellId: string;
605
- profileId: string;
606
- sourceProfile: AgentProfileSource;
607
- harness?: AgentProfileHarness;
608
- model?: string;
609
- promptHash?: string;
610
- dimensions?: Record<string, AgentProfileDimensionValue>;
611
- }
612
-
613
- /**
614
- * Paper-grade RunRecord schema + runtime validator.
615
- *
616
- * Every run that participates in a promotion gate, paper table, or
617
- * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
618
- * fields are exactly those the paper "Two Loops, Three Roles" requires
619
- * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
620
- * holdout split tag. A task score is optional because execution-only records
621
- * must preserve missing labels instead of converting errors into zero quality.
622
- *
623
- * This is intentionally NOT a replacement for the rich `Run` /
624
- * `ProposeReviewReport` / `ScenarioResult` types already in the
625
- * package. Those are runtime structures with full provenance. A
626
- * `RunRecord` is the analysis-time projection — the JSON-friendly
627
- * row you'd put in a parquet file or paste into a notebook.
628
- *
629
- * Validate at the boundary:
630
- *
631
- * const rec = validateRunRecord(rawJson) // throws on missing
632
- * const ok = isRunRecord(rawJson) // boolean check
633
- * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
634
- *
635
- * The validator runs in pure TS — zod is intentionally NOT a
636
- * dependency. Round-trip tested in `tests/run-record.test.ts`.
637
- */
638
-
639
- /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
640
- * combined train+test pool that the optimizer is allowed to read. */
641
- type RunSplitTag = 'search' | 'dev' | 'holdout';
642
- /**
643
- * Explicit execution-lifecycle result for a run.
644
- *
645
- * This is separate from task quality (`outcome`) and failure classification.
646
- * Producers set it only from root-run or process evidence.
647
- */
648
- type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown';
649
- interface RunTokenUsage {
650
- input: number;
651
- /** All generated tokens charged as output, including reasoning tokens. */
652
- output: number;
653
- /** Reasoning-token subset of `output`, when the provider reports it. */
654
- reasoning?: number;
655
- /** Prompt tokens served from a provider cache. */
656
- cached?: number;
657
- /** Prompt tokens written into a provider cache. */
658
- cacheWrite?: number;
659
- }
660
- /**
661
- * How a run's USD amount was obtained.
662
- */
663
- type RunCostProvenance = {
664
- kind: 'observed';
665
- usd: number;
666
- } | {
667
- kind: 'estimated';
668
- usd: number;
669
- } | {
670
- kind: 'uncaptured';
671
- usd: null;
672
- };
673
- interface RunJudgeMetadata {
674
- model: string;
675
- promptVersion: string;
676
- /** [0,1] confidence the judge declared. Constant judge confidence
677
- * across many runs is a fallback signal (see `canary.ts`). */
678
- confidence: number;
679
- /** True if the judge degraded to a fallback path (rules-only,
680
- * prior-call cache, etc.). The canary uses this to alert. */
681
- fallback: boolean;
682
- }
683
- /**
684
- * Per-judge / per-dimension breakdown for runs scored by an ensemble of
685
- * judges over a multi-dimensional rubric.
686
- *
687
- * The collapsed `outcome.searchScore` / `holdoutScore` carries the
688
- * composite the gate uses. The full breakdown belongs here so consumers
689
- * can answer "which judge disagreed?", "which dimension dragged the
690
- * composite down?", and "did half the panel fail?" without re-running.
691
- *
692
- * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
693
- * `composite` are convenience projections — derivable but precomputed so
694
- * downstream IRR primitives (`interRaterReliability`,
695
- * `corpusInterRaterAgreement`) and reporters don't pay the same
696
- * aggregation twice.
697
- *
698
- * Fail-loud discipline: judges that errored out land in `failedJudges`
699
- * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
700
- * run); the explicit list makes a partial-failure recorded as such.
701
- */
702
- interface JudgeScoresRecord {
703
- /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
704
- perJudge: Record<string, Record<string, number>>;
705
- /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
706
- perDimMean: Record<string, number>;
707
- /** Composite mean across successful judges. Mirrors the task score only
708
- * when `failedJudges` is empty. */
709
- composite: number;
710
- /** Judges that errored or returned an unparseable verdict. Recorded
711
- * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
712
- * not inferred from missing keys in `perJudge`. */
713
- failedJudges?: string[];
714
- /** Free-form notes the judges emitted (joined across judges or
715
- * first-judge only — consumer's choice). */
716
- notes?: string;
717
- }
718
- interface RunOutcome {
719
- /** Score on the search/optimization split. Optional for holdout-only and
720
- * execution-only records. */
721
- searchScore?: number;
722
- /** Score on the held-out split. Optional for search-only and execution-only
723
- * records. When both scores are absent, the run is explicitly unlabeled. */
724
- holdoutScore?: number;
725
- /** Bag of any other metric the run produced — judge dimensions,
726
- * pass/fail counters, latency stats, etc. Numeric only — keeps
727
- * reporters honest. */
728
- raw: Record<string, number>;
729
- /** Per-judge / per-dim breakdown. Consumers writing ensemble
730
- * judgements populate this; substrate primitives like
731
- * `interRaterReliability` and `corpusInterRaterAgreement` accept
732
- * these records as input. Optional — single-judge or scalar-only
733
- * runs leave it unset. */
734
- judgeScores?: JudgeScoresRecord;
735
- /** Authenticity / realness verdict — did the run build the REAL thing on the
736
- * intended infra, or fake it (see `./authenticity`)? Optional: only domains
737
- * with an authenticity config populate it. Carried in the corpus so the
738
- * flywheel / off-policy learning can optimize for real completion, not gamed
739
- * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
740
- * must not count as a real success regardless of `score`. */
741
- realness?: {
742
- score: number;
743
- gated: boolean;
744
- reason?: string;
745
- };
746
- }
747
- /**
748
- * Mandatory paper-grade fields for a single evaluation run. Optional
749
- * fields are extension points; mandatory fields throw if missing.
750
- *
751
- * Hash discipline:
752
- * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
753
- * model (after any steering bundle merge).
754
- * - `configHash` is the sha256 of the effective run config (model,
755
- * temperature, tools, judges, splits). The pair (promptHash,
756
- * configHash) uniquely identifies an experiment cell.
757
- *
758
- * Model snapshot discipline:
759
- * - `model` MUST encode a snapshot version. Bare aliases like
760
- * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
761
- * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
762
- */
763
- interface RunRecord {
764
- /** UUID for the run. */
765
- runId: string;
766
- /** Logical experiment grouping (a treatment vs a baseline within
767
- * the same sweep should share `experimentId`). */
768
- experimentId: string;
769
- /** Stable identifier for the candidate (variant) being run. The
770
- * promotion gate compares two `candidateId`s on matched items. */
771
- candidateId: string;
772
- /** RNG seed for the run. Always recorded — silent re-seeding is
773
- * the most common cause of non-reproducible numbers. */
774
- seed: number;
775
- /** Model identifier WITH snapshot version. */
776
- model: string;
777
- /** sha256 of the effective prompt (post-steering). */
778
- promptHash: string;
779
- /** sha256 of the effective config. */
780
- configHash: string;
781
- /** Git SHA the harness was run from. */
782
- commitSha: string;
783
- /** End-to-end wall-clock duration in milliseconds. */
784
- wallMs: number;
785
- /** Time spent queued before execution started, if known. */
786
- queueMs?: number;
787
- /** Total USD cost, or null when the producer could not capture one. */
788
- costUsd: number | null;
789
- /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */
790
- costProvenance: RunCostProvenance;
791
- /** Token usage breakdown. */
792
- tokenUsage: RunTokenUsage;
793
- /** Root-run or process terminal result. Never inferred from a child span. */
794
- terminalOutcome: RunTerminalOutcome;
795
- /** Root-run or process failure reason. Valid only for a failed, cancelled,
796
- * or incomplete terminal result; never populated from a child span. */
797
- terminalFailureReason?: string;
798
- /** Judge-side metadata, if a judge was used. */
799
- judgeMetadata?: RunJudgeMetadata;
800
- /** Per-split scores + raw bag. */
801
- outcome: RunOutcome;
802
- /** Canonical task-failure class drawn from the shared
803
- * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result
804
- * evidence. Execution errors belong in
805
- * `outcome.raw.execution_error_count`. */
806
- failureClass?: FailureClass;
807
- /** Free-form task-failure detail scoped under a non-success
808
- * `failureClass`. It is invalid without that class. */
809
- failureMode?: string;
810
- /** Which split this run was drawn from. */
811
- splitTag: RunSplitTag;
812
- /**
813
- * Stable scenario identifier the run observed or was scored against.
814
- * Comparison primitives match this identity rather than input order.
815
- */
816
- scenarioId: string;
817
- /**
818
- * Canonical identity for the agent profile cell that produced this row:
819
- * profile artifact hash plus optional harness/model/prompt/reporting
820
- * dimensions. Use `agentProfile.cellId` to group persona sweeps and
821
- * longitudinal reports by the complete source profile, not by a loose
822
- * candidate label or opaque config hash.
823
- */
824
- agentProfile?: AgentProfileCell;
825
- }
826
-
827
- type CodeAgentSessionSource = 'codex' | 'claude-code' | 'opencode' | 'kimi-code' | 'pi';
828
- type CodeAgentSessionTerminalStatus = 'completed' | 'failed' | 'unknown';
829
- type CodeAgentSessionActionKind = 'tool' | 'patch' | 'terminal' | 'graph-completion';
830
- type CodeAgentSessionActionSurface = 'tool' | 'mcp' | 'subagent' | 'skill' | 'hook' | 'web' | 'code';
831
- type CodeAgentSessionActionStatus = 'started' | 'completed' | 'failed' | 'unknown';
832
- interface CodeAgentSessionExecutionReceipt {
833
- exitCode: number;
834
- startedAtMs?: number;
835
- completedAtMs?: number;
836
- }
837
- interface CodeAgentSessionAction {
838
- id: string;
839
- stepIndex: number;
840
- kind: CodeAgentSessionActionKind;
841
- surface: CodeAgentSessionActionSurface;
842
- name: string;
843
- status: CodeAgentSessionActionStatus;
844
- timestampMs?: number;
845
- costUsd?: number;
846
- metadata: Record<string, unknown>;
847
- }
848
- interface CodeAgentSessionObservation {
849
- source: CodeAgentSessionSource;
850
- sessionId: string;
851
- finalText: string | null;
852
- terminal: {
853
- status: CodeAgentSessionTerminalStatus;
854
- explicit: boolean;
855
- };
856
- actions: CodeAgentSessionAction[];
857
- }
858
-
859
- interface CodeAgentSessionMetrics {
860
- entries: number;
861
- userMessages: number;
862
- assistantMessages: number;
863
- reasoningItems: number;
864
- toolCalls: number;
865
- toolOutputs: number;
866
- toolErrors: number;
867
- unclassifiedErrors: number;
868
- patchAttempts: number;
869
- patchSuccesses: number;
870
- patchFailures: number;
871
- turnsStarted: number;
872
- turnsCompleted: number;
873
- turnsAborted: number;
874
- contextCompactions: number;
875
- mcpCalls: number;
876
- subagentCalls: number;
877
- skillCalls: number;
878
- hookCalls: number;
879
- webCalls: number;
880
- codeActions: number;
881
- prLinks: number;
882
- fileSnapshots: number;
883
- graphNodes: number;
884
- graphEdges: number;
885
- actionCandidates: number;
886
- verificationReports: number;
887
- completionDecisions: number;
888
- reliabilityRows: number;
889
- reliabilityLift: number;
890
- inputTokens: number;
891
- outputTokens: number;
892
- reasoningTokens: number;
893
- cachedTokens: number;
894
- cacheWriteTokens: number;
895
- observedCostUsd: number;
896
- observedCostCaptured?: boolean;
897
- wallMs: number;
898
- processScore: number;
899
- }
900
- interface CodeAgentSessionDiagnostic {
901
- source: CodeAgentSessionSource;
902
- sessionId: string;
903
- sourcePath?: string;
904
- entries: number;
905
- malformedLines: number;
906
- hasExplicitTerminalSignal: boolean;
907
- hasFinalOutput: boolean;
908
- hasQualityLabel: boolean;
909
- hasTokenUsage: boolean;
910
- hasCost: boolean;
911
- costKind?: RunCostProvenance['kind'];
912
- warnings: string[];
913
- }
914
- interface CodeAgentSessionIntakeOptions {
915
- entries: unknown[];
916
- malformedLines?: number;
917
- sourcePath?: string;
918
- experimentId?: string;
919
- candidateId?: string;
920
- seed?: number;
921
- splitTag?: RunSplitTag;
922
- scenarioId?: string;
923
- model?: string;
924
- promptHash?: string;
925
- configHash?: string;
926
- commitSha?: string;
927
- score?: number;
928
- /** Explicit cost receipt. When omitted, source-reported cost wins, then a
929
- * token-priced estimate, then uncaptured. */
930
- costProvenance?: RunCostProvenance;
931
- /** Exact executor-owned process result. This is required when a provider's
932
- * JSON stream has no terminal event, as with `opencode run --format json`. */
933
- execution?: CodeAgentSessionExecutionReceipt;
934
- }
935
-
213
+ //#endregion
214
+ //#region src/belief-state/ope.d.ts
936
215
  interface BeliefOpeOptions extends OffPolicyOptions {
937
- minEffectiveSampleSize?: number;
938
- minEffectiveSampleRatio?: number;
939
- maxDiagnostics?: number;
216
+ minEffectiveSampleSize?: number;
217
+ minEffectiveSampleRatio?: number;
218
+ maxDiagnostics?: number;
940
219
  }
941
220
  interface BeliefOffPolicyTrajectoryReport {
942
- targetPolicyId: string;
943
- trajectories: OffPolicyTrajectory[];
944
- dropped: number;
945
- diagnostics: string[];
221
+ targetPolicyId: string;
222
+ trajectories: OffPolicyTrajectory[];
223
+ dropped: number;
224
+ diagnostics: string[];
946
225
  }
947
226
  declare function embeddedBeliefOpeTargetPolicy(id?: string): BeliefOpeTargetPolicy;
948
227
  declare function beliefDecisionsToOffPolicyTrajectories(points: BeliefDecisionPoint[], targetPolicy: BeliefOpeTargetPolicy, options?: Pick<BeliefOpeOptions, 'maxDiagnostics'>): BeliefOffPolicyTrajectoryReport;
949
228
  declare function evaluateBeliefOffPolicy(points: BeliefDecisionPoint[], targetPolicy: BeliefOpeTargetPolicy, options?: BeliefOpeOptions): BeliefOpeReport;
950
-
229
+ //#endregion
230
+ //#region src/belief-state/selective.d.ts
951
231
  interface EvaluateBeliefSelectivePolicyOptions {
952
- utility?: BeliefUtilityOptions;
953
- minN?: number;
954
- minAccepted?: number;
955
- minUtilityDelta?: number;
956
- seed?: number;
232
+ utility?: BeliefUtilityOptions;
233
+ minN?: number;
234
+ minAccepted?: number;
235
+ minUtilityDelta?: number;
236
+ seed?: number;
957
237
  }
958
238
  declare function thresholdSelectivePolicy(options: {
959
- id?: string;
960
- confidenceThreshold: number;
961
- belowThresholdAction?: Exclude<BeliefPolicyAction, 'accept'>;
239
+ id?: string;
240
+ confidenceThreshold: number;
241
+ belowThresholdAction?: Exclude<BeliefPolicyAction, 'accept'>;
962
242
  }): BeliefSelectivePolicy;
963
243
  declare function evaluateBeliefSelectivePolicy(points: BeliefDecisionPoint[], policy: BeliefSelectivePolicy, options?: EvaluateBeliefSelectivePolicyOptions): BeliefSelectivePolicyMetrics;
964
-
244
+ //#endregion
245
+ //#region src/belief-state/report.d.ts
965
246
  interface AnalyzeBeliefPolicyOpeOptions extends BeliefOpeOptions {
966
- targetPolicy?: BeliefOpeTargetPolicy;
247
+ targetPolicy?: BeliefOpeTargetPolicy;
967
248
  }
968
249
  interface AnalyzeBeliefPolicyOptions {
969
- points: BeliefDecisionPoint[];
970
- policy: BeliefSelectivePolicy;
971
- selective?: EvaluateBeliefSelectivePolicyOptions;
972
- calibration?: BeliefCalibrationOptions;
973
- ope?: AnalyzeBeliefPolicyOpeOptions;
974
- requireOpe?: boolean;
250
+ points: BeliefDecisionPoint[];
251
+ policy: BeliefSelectivePolicy;
252
+ selective?: EvaluateBeliefSelectivePolicyOptions;
253
+ calibration?: BeliefCalibrationOptions;
254
+ ope?: AnalyzeBeliefPolicyOpeOptions;
255
+ requireOpe?: boolean;
975
256
  }
976
257
  declare function analyzeBeliefPolicy(options: AnalyzeBeliefPolicyOptions): BeliefPolicyEvaluationReport;
977
-
258
+ //#endregion
259
+ //#region src/belief-state/code-agent-corpus.d.ts
978
260
  type CodeAgentBeliefDecisionTargetId = 'failure-recovery' | 'tool-selection' | 'graph-completion';
979
261
  interface ExtractCodeAgentBeliefDecisionPointsOptions {
980
- source: CodeAgentSessionSource;
981
- entries: unknown[];
982
- /** Reuse intake's provider-neutral projection when available. */
983
- observation?: CodeAgentSessionObservation;
984
- run: Pick<RunRecord, 'runId' | 'scenarioId' | 'outcome' | 'costUsd'>;
985
- sourcePath?: string;
262
+ source: CodeAgentSessionSource;
263
+ entries: unknown[];
264
+ /** Reuse intake's provider-neutral projection when available. */
265
+ observation?: CodeAgentSessionObservation;
266
+ run: Pick<RunRecord, 'runId' | 'scenarioId' | 'outcome' | 'costUsd'>;
267
+ sourcePath?: string;
986
268
  }
987
269
  interface BeliefDecisionInventoryBucket {
988
- id: string;
989
- kind?: BeliefDecisionKind;
990
- targetId?: CodeAgentBeliefDecisionTargetId;
991
- n: number;
992
- withOutcome: number;
993
- withConfidence: number;
994
- withCandidateActions: number;
995
- withBehaviorProb: number;
996
- withTargetProb: number;
997
- successRate: number | null;
998
- meanScore: number | null;
999
- meanConfidence: number | null;
270
+ id: string;
271
+ kind?: BeliefDecisionKind;
272
+ targetId?: CodeAgentBeliefDecisionTargetId;
273
+ n: number;
274
+ withOutcome: number;
275
+ withConfidence: number;
276
+ withCandidateActions: number;
277
+ withBehaviorProb: number;
278
+ withTargetProb: number;
279
+ successRate: number | null;
280
+ meanScore: number | null;
281
+ meanConfidence: number | null;
1000
282
  }
1001
283
  interface BeliefDecisionInventoryReport {
1002
- n: number;
1003
- byKind: BeliefDecisionInventoryBucket[];
1004
- byTarget: BeliefDecisionInventoryBucket[];
1005
- diagnostics: string[];
284
+ n: number;
285
+ byKind: BeliefDecisionInventoryBucket[];
286
+ byTarget: BeliefDecisionInventoryBucket[];
287
+ diagnostics: string[];
1006
288
  }
1007
289
  interface BeliefDecisionTargetSelection {
1008
- id: CodeAgentBeliefDecisionTargetId;
1009
- label: string;
1010
- points: BeliefDecisionPoint[];
1011
- support: BeliefDecisionInventoryBucket;
1012
- reasons: string[];
290
+ id: CodeAgentBeliefDecisionTargetId;
291
+ label: string;
292
+ points: BeliefDecisionPoint[];
293
+ support: BeliefDecisionInventoryBucket;
294
+ reasons: string[];
1013
295
  }
1014
296
  interface SelectBeliefDecisionTargetOptions {
1015
- minN?: number;
1016
- minOutcomeCoverage?: number;
1017
- preferredTargets?: CodeAgentBeliefDecisionTargetId[];
297
+ minN?: number;
298
+ minOutcomeCoverage?: number;
299
+ preferredTargets?: CodeAgentBeliefDecisionTargetId[];
1018
300
  }
1019
301
  interface AnalyzeBeliefDecisionCorpusOptions {
1020
- points: BeliefDecisionPoint[];
1021
- targetId?: CodeAgentBeliefDecisionTargetId;
1022
- minN?: number;
1023
- minOutcomeCoverage?: number;
1024
- minAccepted?: number;
1025
- confidenceThreshold?: number;
1026
- policy?: BeliefSelectivePolicy;
1027
- requireOpe?: boolean;
1028
- policyOptions?: Partial<AnalyzeBeliefPolicyOptions>;
302
+ points: BeliefDecisionPoint[];
303
+ targetId?: CodeAgentBeliefDecisionTargetId;
304
+ minN?: number;
305
+ minOutcomeCoverage?: number;
306
+ minAccepted?: number;
307
+ confidenceThreshold?: number;
308
+ policy?: BeliefSelectivePolicy;
309
+ requireOpe?: boolean;
310
+ policyOptions?: Partial<AnalyzeBeliefPolicyOptions>;
1029
311
  }
1030
312
  interface BeliefDecisionCorpusEvaluation {
1031
- inventory: BeliefDecisionInventoryReport;
1032
- target?: BeliefDecisionTargetSelection;
1033
- policy?: BeliefSelectivePolicy;
1034
- evaluation?: BeliefPolicyEvaluationReport;
1035
- diagnostics: string[];
313
+ inventory: BeliefDecisionInventoryReport;
314
+ target?: BeliefDecisionTargetSelection;
315
+ policy?: BeliefSelectivePolicy;
316
+ evaluation?: BeliefPolicyEvaluationReport;
317
+ diagnostics: string[];
1036
318
  }
1037
319
  declare function extractCodeAgentBeliefDecisionPoints(options: ExtractCodeAgentBeliefDecisionPointsOptions): BeliefDecisionExtractionReport;
1038
320
  declare function inventoryBeliefDecisionPoints(points: BeliefDecisionPoint[]): BeliefDecisionInventoryReport;
1039
321
  declare function selectBeliefDecisionTarget(points: BeliefDecisionPoint[], options?: SelectBeliefDecisionTargetOptions): BeliefDecisionTargetSelection | null;
1040
322
  declare function analyzeBeliefDecisionCorpus(options: AnalyzeBeliefDecisionCorpusOptions): BeliefDecisionCorpusEvaluation;
1041
-
323
+ //#endregion
324
+ //#region src/belief-state/research-evidence.d.ts
1042
325
  type BeliefResearchClaimScope = 'selective' | 'counterfactual';
1043
326
  type BeliefResearchEvidenceStatus = 'supported' | 'blocked';
1044
327
  type BeliefResearchGateId = 'corpus' | 'selective' | 'calibration' | 'ope';
1045
328
  interface BeliefResearchEvidenceGate {
1046
- id: BeliefResearchGateId;
1047
- status: BeliefResearchEvidenceStatus;
1048
- blockers: string[];
1049
- caveats: string[];
329
+ id: BeliefResearchGateId;
330
+ status: BeliefResearchEvidenceStatus;
331
+ blockers: string[];
332
+ caveats: string[];
1050
333
  }
1051
334
  interface BeliefDecisionResearchEvidencePacket {
1052
- claimScope: BeliefResearchClaimScope;
1053
- status: BeliefResearchEvidenceStatus;
1054
- analysis: BeliefDecisionCorpusEvaluation;
1055
- gates: BeliefResearchEvidenceGate[];
1056
- blockers: string[];
1057
- caveats: string[];
335
+ claimScope: BeliefResearchClaimScope;
336
+ status: BeliefResearchEvidenceStatus;
337
+ analysis: BeliefDecisionCorpusEvaluation;
338
+ gates: BeliefResearchEvidenceGate[];
339
+ blockers: string[];
340
+ caveats: string[];
1058
341
  }
1059
342
  interface BuildBeliefDecisionResearchEvidencePacketOptions extends AnalyzeBeliefDecisionCorpusOptions {
1060
- claimScope?: BeliefResearchClaimScope;
343
+ claimScope?: BeliefResearchClaimScope;
1061
344
  }
1062
345
  declare function buildBeliefDecisionResearchEvidencePacket(options: BuildBeliefDecisionResearchEvidencePacketOptions): BeliefDecisionResearchEvidencePacket;
1063
-
346
+ //#endregion
347
+ //#region src/belief-state/code-agent-evidence.d.ts
1064
348
  interface CodeAgentBeliefSession extends CodeAgentSessionIntakeOptions {
1065
- source: CodeAgentSessionSource;
349
+ source: CodeAgentSessionSource;
1066
350
  }
1067
351
  interface BuildCodeAgentBeliefEvidenceCorpusOptions extends Omit<BuildBeliefDecisionResearchEvidencePacketOptions, 'points'> {
1068
- sessions: CodeAgentBeliefSession[];
352
+ sessions: CodeAgentBeliefSession[];
1069
353
  }
1070
354
  interface CodeAgentBeliefEvidenceCorpus {
1071
- runs: RunRecord[];
1072
- metrics: CodeAgentSessionMetrics[];
1073
- intakeDiagnostics: CodeAgentSessionDiagnostic[];
1074
- extractionDiagnostics: BeliefDecisionExtractionDiagnostic[];
1075
- decisions: BeliefDecisionPoint[];
1076
- inventory: BeliefDecisionInventoryReport;
1077
- evidence: BeliefDecisionResearchEvidencePacket;
355
+ runs: RunRecord[];
356
+ metrics: CodeAgentSessionMetrics[];
357
+ intakeDiagnostics: CodeAgentSessionDiagnostic[];
358
+ extractionDiagnostics: BeliefDecisionExtractionDiagnostic[];
359
+ decisions: BeliefDecisionPoint[];
360
+ inventory: BeliefDecisionInventoryReport;
361
+ evidence: BeliefDecisionResearchEvidencePacket;
1078
362
  }
1079
363
  declare function buildCodeAgentBeliefEvidenceCorpus(options: BuildCodeAgentBeliefEvidenceCorpusOptions): CodeAgentBeliefEvidenceCorpus;
1080
-
364
+ //#endregion
365
+ //#region src/belief-state/extract.d.ts
1081
366
  interface ExtractBeliefDecisionPointsOptions {
1082
- runIds?: string[];
367
+ runIds?: string[];
1083
368
  }
1084
369
  declare function extractBeliefDecisionPoints(store: TraceStore, options?: ExtractBeliefDecisionPointsOptions): Promise<BeliefDecisionExtractionReport>;
1085
-
370
+ //#endregion
371
+ //#region src/belief-state/shadow-probe.d.ts
1086
372
  interface BeliefShadowProbeInput {
1087
- probeId: string;
1088
- decisionId: string;
1089
- runId: string;
1090
- scenarioId?: string;
1091
- stepIndex: number;
1092
- decisionKind: BeliefDecisionKind;
1093
- candidateActions: string[];
1094
- observedAction?: string;
1095
- evidence: BeliefShadowProbeEvidenceRef[];
1096
- context?: string;
1097
- metadata?: Record<string, unknown>;
373
+ probeId: string;
374
+ decisionId: string;
375
+ runId: string;
376
+ scenarioId?: string;
377
+ stepIndex: number;
378
+ decisionKind: BeliefDecisionKind;
379
+ candidateActions: string[];
380
+ observedAction?: string;
381
+ evidence: BeliefShadowProbeEvidenceRef[];
382
+ context?: string;
383
+ metadata?: Record<string, unknown>;
1098
384
  }
1099
385
  interface BeliefShadowProbeEvidenceRef {
1100
- id: string;
1101
- source: string;
1102
- detail?: string;
1103
- quality?: BeliefEvidenceQuality;
386
+ id: string;
387
+ source: string;
388
+ detail?: string;
389
+ quality?: BeliefEvidenceQuality;
1104
390
  }
1105
391
  interface BeliefShadowProbeResponse {
1106
- predictedAction: string;
1107
- confidence: number;
1108
- beliefSummary?: string;
1109
- uncertainty?: string[];
1110
- evidenceRefs?: string[];
1111
- wouldChangeMindIf?: string[];
1112
- targetProb?: number;
1113
- qHat?: number | null;
1114
- metadata?: Record<string, unknown>;
392
+ predictedAction: string;
393
+ confidence: number;
394
+ beliefSummary?: string;
395
+ uncertainty?: string[];
396
+ evidenceRefs?: string[];
397
+ wouldChangeMindIf?: string[];
398
+ targetProb?: number;
399
+ qHatChosen?: number | null;
400
+ vHatTarget?: number | null;
401
+ metadata?: Record<string, unknown>;
1115
402
  }
1116
403
  interface BeliefShadowProbeRecord extends BeliefShadowProbeResponse {
1117
- probeId: string;
1118
- decisionId: string;
1119
- runId: string;
1120
- scenarioId?: string;
1121
- stepIndex: number;
1122
- decisionKind: BeliefDecisionKind;
1123
- candidateActions: string[];
1124
- observedAction: string;
1125
- agreesWithObservedAction: boolean;
1126
- outcome?: BeliefDecisionOutcome;
404
+ probeId: string;
405
+ decisionId: string;
406
+ runId: string;
407
+ scenarioId?: string;
408
+ stepIndex: number;
409
+ decisionKind: BeliefDecisionKind;
410
+ candidateActions: string[];
411
+ observedAction: string;
412
+ agreesWithObservedAction: boolean;
413
+ outcome?: BeliefDecisionOutcome;
1127
414
  }
1128
415
  interface BeliefShadowProbeDiagnostic {
1129
- decisionId: string;
1130
- severity: 'warning' | 'error';
1131
- reason: string;
416
+ decisionId: string;
417
+ severity: 'warning' | 'error';
418
+ reason: string;
1132
419
  }
1133
420
  interface BeliefShadowProbeSummary {
1134
- attempted: number;
1135
- completed: number;
1136
- dropped: number;
1137
- withOutcome: number;
1138
- withTargetProb: number;
1139
- meanConfidence: number | null;
1140
- observedAgreementRate: number | null;
421
+ attempted: number;
422
+ completed: number;
423
+ dropped: number;
424
+ withOutcome: number;
425
+ withTargetProb: number;
426
+ meanConfidence: number | null;
427
+ observedAgreementRate: number | null;
1141
428
  }
1142
429
  interface BeliefShadowProbeRun {
1143
- probeId: string;
1144
- records: BeliefShadowProbeRecord[];
1145
- diagnostics: BeliefShadowProbeDiagnostic[];
1146
- summary: BeliefShadowProbeSummary;
430
+ probeId: string;
431
+ records: BeliefShadowProbeRecord[];
432
+ diagnostics: BeliefShadowProbeDiagnostic[];
433
+ summary: BeliefShadowProbeSummary;
1147
434
  }
1148
435
  interface RunBeliefShadowProbeOptions {
1149
- probeId: string;
1150
- points: BeliefDecisionPoint[];
1151
- probe: (input: BeliefShadowProbeInput) => BeliefShadowProbeResponse | Promise<BeliefShadowProbeResponse>;
1152
- contextOf?: (point: BeliefDecisionPoint) => string | undefined | Promise<string | undefined>;
1153
- metadataOf?: (point: BeliefDecisionPoint) => Record<string, unknown> | undefined | Promise<Record<string, unknown> | undefined>;
1154
- includeObservedAction?: boolean;
1155
- includeEvidenceDetail?: boolean;
1156
- includeOutcomeInRecord?: boolean;
1157
- requireCandidateActions?: boolean;
1158
- allowOutOfSetActions?: boolean;
1159
- concurrency?: number;
1160
- maxContextChars?: number;
436
+ probeId: string;
437
+ points: BeliefDecisionPoint[];
438
+ probe: (input: BeliefShadowProbeInput) => BeliefShadowProbeResponse | Promise<BeliefShadowProbeResponse>;
439
+ contextOf?: (point: BeliefDecisionPoint) => string | undefined | Promise<string | undefined>;
440
+ metadataOf?: (point: BeliefDecisionPoint) => Record<string, unknown> | undefined | Promise<Record<string, unknown> | undefined>;
441
+ includeObservedAction?: boolean;
442
+ includeEvidenceDetail?: boolean;
443
+ includeOutcomeInRecord?: boolean;
444
+ requireCandidateActions?: boolean;
445
+ allowOutOfSetActions?: boolean;
446
+ concurrency?: number;
447
+ maxContextChars?: number;
1161
448
  }
1162
449
  declare function runBeliefShadowProbe(options: RunBeliefShadowProbeOptions): Promise<BeliefShadowProbeRun>;
1163
450
  declare function formatBeliefShadowProbePrompt(input: BeliefShadowProbeInput): string;
1164
-
451
+ //#endregion
452
+ //#region src/belief-state/runtime-hooks.d.ts
1165
453
  interface RuntimeBeliefDecisionEvidenceRef {
1166
- source: string;
1167
- id: string;
1168
- detail?: string;
1169
- quality?: BeliefEvidenceQuality;
1170
- metadata?: Record<string, unknown>;
454
+ source: string;
455
+ id: string;
456
+ detail?: string;
457
+ quality?: BeliefEvidenceQuality;
458
+ metadata?: Record<string, unknown>;
1171
459
  }
1172
460
  interface RuntimeBeliefDecisionPoint {
1173
- id: string;
1174
- runId: string;
1175
- scenarioId?: string;
1176
- stepIndex: number;
1177
- kind: string;
1178
- candidateActions?: string[];
1179
- context?: string;
1180
- evidence?: RuntimeBeliefDecisionEvidenceRef[];
1181
- metadata?: Record<string, unknown>;
461
+ id: string;
462
+ runId: string;
463
+ scenarioId?: string;
464
+ stepIndex: number;
465
+ kind: string;
466
+ candidateActions?: string[];
467
+ context?: string;
468
+ evidence?: RuntimeBeliefDecisionEvidenceRef[];
469
+ metadata?: Record<string, unknown>;
1182
470
  }
1183
471
  interface RuntimeBeliefHookEvent {
1184
- id: string;
1185
- runId: string;
1186
- scenarioId?: string;
1187
- target: string;
1188
- phase: string;
1189
- timestamp: number;
1190
- stepIndex?: number;
1191
- parentId?: string;
1192
- payload?: unknown;
1193
- metadata?: Record<string, unknown>;
472
+ id: string;
473
+ runId: string;
474
+ scenarioId?: string;
475
+ target: string;
476
+ phase: string;
477
+ timestamp: number;
478
+ stepIndex?: number;
479
+ parentId?: string;
480
+ payload?: unknown;
481
+ metadata?: Record<string, unknown>;
1194
482
  }
1195
483
  interface RuntimeBeliefHookContext {
1196
- signal?: AbortSignal;
484
+ signal?: AbortSignal;
1197
485
  }
1198
486
  interface RuntimeBeliefHooks {
1199
- onEvent?: (event: RuntimeBeliefHookEvent, context: RuntimeBeliefHookContext) => void | Promise<void>;
1200
- onDecisionPoint?: (point: RuntimeBeliefDecisionPoint, context: RuntimeBeliefHookContext) => void | Promise<void>;
487
+ onEvent?: (event: RuntimeBeliefHookEvent, context: RuntimeBeliefHookContext) => void | Promise<void>;
488
+ onDecisionPoint?: (point: RuntimeBeliefDecisionPoint, context: RuntimeBeliefHookContext) => void | Promise<void>;
1201
489
  }
1202
490
  interface RuntimeBeliefConversionDiagnostic {
1203
- decisionId: string;
1204
- severity: 'warning' | 'error';
1205
- reason: string;
491
+ decisionId: string;
492
+ severity: 'warning' | 'error';
493
+ reason: string;
1206
494
  }
1207
495
  interface RuntimeBeliefShadowProbeInputOptions {
1208
- probeId: string;
1209
- decisionKind?: BeliefDecisionKind;
1210
- includeEvidenceDetail?: boolean;
1211
- includeLifecycleEvidence?: boolean;
1212
- lifecycleEvents?: RuntimeBeliefHookEvent[];
1213
- maxContextChars?: number;
496
+ probeId: string;
497
+ decisionKind?: BeliefDecisionKind;
498
+ includeEvidenceDetail?: boolean;
499
+ includeLifecycleEvidence?: boolean;
500
+ lifecycleEvents?: RuntimeBeliefHookEvent[];
501
+ maxContextChars?: number;
1214
502
  }
1215
503
  interface RuntimeBeliefDecisionPointOptions {
1216
- chosenAction?: string;
1217
- decisionKind?: BeliefDecisionKind;
1218
- confidence?: number;
1219
- behaviorProb?: number;
1220
- targetProb?: number;
1221
- qHatChosen?: number | null;
1222
- vHatTarget?: number | null;
1223
- /** @deprecated Use `qHatChosen` and `vHatTarget` together. */
1224
- qHat?: number | null;
1225
- costUsd?: number;
1226
- outcome?: BeliefDecisionOutcome;
1227
- metadata?: Record<string, unknown>;
1228
- includeLifecycleEvidence?: boolean;
1229
- lifecycleEvents?: RuntimeBeliefHookEvent[];
504
+ chosenAction?: string;
505
+ decisionKind?: BeliefDecisionKind;
506
+ confidence?: number;
507
+ behaviorProb?: number;
508
+ targetProb?: number;
509
+ qHatChosen?: number | null;
510
+ vHatTarget?: number | null;
511
+ costUsd?: number;
512
+ outcome?: BeliefDecisionOutcome;
513
+ metadata?: Record<string, unknown>;
514
+ includeLifecycleEvidence?: boolean;
515
+ lifecycleEvents?: RuntimeBeliefHookEvent[];
1230
516
  }
1231
517
  interface RuntimeBeliefShadowProbeInputReport {
1232
- input?: BeliefShadowProbeInput;
1233
- diagnostics: RuntimeBeliefConversionDiagnostic[];
518
+ input?: BeliefShadowProbeInput;
519
+ diagnostics: RuntimeBeliefConversionDiagnostic[];
1234
520
  }
1235
521
  interface RuntimeBeliefDecisionPointReport {
1236
- point?: BeliefDecisionPoint;
1237
- diagnostics: RuntimeBeliefConversionDiagnostic[];
522
+ point?: BeliefDecisionPoint;
523
+ diagnostics: RuntimeBeliefConversionDiagnostic[];
1238
524
  }
1239
525
  interface BeliefRuntimeHookCollector {
1240
- hooks: RuntimeBeliefHooks;
1241
- decisions: RuntimeBeliefDecisionPoint[];
1242
- events: RuntimeBeliefHookEvent[];
1243
- toShadowProbeInputs(options?: Partial<RuntimeBeliefShadowProbeInputOptions>): {
1244
- inputs: BeliefShadowProbeInput[];
1245
- diagnostics: RuntimeBeliefConversionDiagnostic[];
1246
- };
1247
- clear(): void;
526
+ hooks: RuntimeBeliefHooks;
527
+ decisions: RuntimeBeliefDecisionPoint[];
528
+ events: RuntimeBeliefHookEvent[];
529
+ toShadowProbeInputs(options?: Partial<RuntimeBeliefShadowProbeInputOptions>): {
530
+ inputs: BeliefShadowProbeInput[];
531
+ diagnostics: RuntimeBeliefConversionDiagnostic[];
532
+ };
533
+ clear(): void;
1248
534
  }
1249
535
  declare function runtimeDecisionPointToBeliefShadowProbeInput(point: RuntimeBeliefDecisionPoint, options: RuntimeBeliefShadowProbeInputOptions): RuntimeBeliefShadowProbeInputReport;
1250
536
  declare function runtimeDecisionPointToBeliefDecisionPoint(point: RuntimeBeliefDecisionPoint, options: RuntimeBeliefDecisionPointOptions): RuntimeBeliefDecisionPointReport;
1251
537
  declare function createBeliefRuntimeHookCollector(defaults: RuntimeBeliefShadowProbeInputOptions): BeliefRuntimeHookCollector;
1252
-
538
+ //#endregion
539
+ //#region src/belief-state/phase0-measurement.d.ts
1253
540
  interface RuntimeBeliefPhase0RunRecord {
1254
- runId: string;
1255
- scenarioId?: string;
1256
- splitTag: RunSplitTag;
541
+ runId: string;
542
+ scenarioId?: string;
543
+ splitTag: RunSplitTag;
1257
544
  }
1258
545
  interface RuntimeBeliefDecisionLabel {
1259
- decisionId: string;
1260
- chosenAction: string;
1261
- outcome: BeliefDecisionOutcome;
1262
- confidence?: number;
1263
- behaviorProb?: number;
1264
- targetProb?: number;
1265
- qHatChosen?: number | null;
1266
- vHatTarget?: number | null;
1267
- /** @deprecated Use `qHatChosen` and `vHatTarget` together. */
1268
- qHat?: number | null;
1269
- costUsd?: number;
1270
- splitTag?: RunSplitTag;
1271
- metadata?: Record<string, unknown>;
546
+ decisionId: string;
547
+ chosenAction: string;
548
+ outcome: BeliefDecisionOutcome;
549
+ confidence?: number;
550
+ behaviorProb?: number;
551
+ targetProb?: number;
552
+ qHatChosen?: number | null;
553
+ vHatTarget?: number | null;
554
+ costUsd?: number;
555
+ splitTag?: RunSplitTag;
556
+ metadata?: Record<string, unknown>;
1272
557
  }
1273
558
  interface BuildRuntimeBeliefPhase0MeasurementOptions extends Omit<BuildBeliefDecisionResearchEvidencePacketOptions, 'points'> {
1274
- runs: RuntimeBeliefPhase0RunRecord[];
1275
- decisions: RuntimeBeliefDecisionPoint[];
1276
- events?: RuntimeBeliefHookEvent[];
1277
- labels: RuntimeBeliefDecisionLabel[];
1278
- baselinePolicyId?: string;
559
+ runs: RuntimeBeliefPhase0RunRecord[];
560
+ decisions: RuntimeBeliefDecisionPoint[];
561
+ events?: RuntimeBeliefHookEvent[];
562
+ labels: RuntimeBeliefDecisionLabel[];
563
+ baselinePolicyId?: string;
1279
564
  }
1280
565
  interface RuntimeBeliefPhase0MeasurementSummary {
1281
- runCount: number;
1282
- producerDecisionCount: number;
1283
- lifecycleEventCount: number;
1284
- labelCount: number;
1285
- completedPointCount: number;
1286
- runJoinRate: number;
1287
- labelJoinRate: number;
1288
- missingRunRecordCount: number;
1289
- missingLabelCount: number;
1290
- withEvidence: number;
1291
- withOutcome: number;
1292
- withSplit: number;
1293
- withBehaviorProb: number;
1294
- withTargetProb: number;
1295
- baselinePolicyId: string;
1296
- packetStatus: BeliefDecisionResearchEvidencePacket['status'];
1297
- claimScope: BeliefDecisionResearchEvidencePacket['claimScope'];
566
+ runCount: number;
567
+ producerDecisionCount: number;
568
+ lifecycleEventCount: number;
569
+ labelCount: number;
570
+ completedPointCount: number;
571
+ runJoinRate: number;
572
+ labelJoinRate: number;
573
+ missingRunRecordCount: number;
574
+ missingLabelCount: number;
575
+ withEvidence: number;
576
+ withOutcome: number;
577
+ withSplit: number;
578
+ withBehaviorProb: number;
579
+ withTargetProb: number;
580
+ baselinePolicyId: string;
581
+ packetStatus: BeliefDecisionResearchEvidencePacket['status'];
582
+ claimScope: BeliefDecisionResearchEvidencePacket['claimScope'];
1298
583
  }
1299
584
  interface RuntimeBeliefPhase0Measurement {
1300
- points: BeliefDecisionPoint[];
1301
- packet: BeliefDecisionResearchEvidencePacket;
1302
- summary: RuntimeBeliefPhase0MeasurementSummary;
1303
- diagnostics: string[];
585
+ points: BeliefDecisionPoint[];
586
+ packet: BeliefDecisionResearchEvidencePacket;
587
+ summary: RuntimeBeliefPhase0MeasurementSummary;
588
+ diagnostics: string[];
1304
589
  }
1305
590
  declare function buildRuntimeBeliefPhase0Measurement(options: BuildRuntimeBeliefPhase0MeasurementOptions): RuntimeBeliefPhase0Measurement;
1306
-
1307
- interface RuntimeTrajectoryHookEvent {
1308
- id: string;
1309
- runId: string;
1310
- scenarioId?: string;
1311
- target: string;
1312
- phase: string;
1313
- timestamp: number;
1314
- stepIndex?: number;
1315
- parentId?: string;
1316
- payload?: unknown;
1317
- metadata?: Record<string, unknown>;
1318
- }
1319
- interface RuntimeTrajectoryRecord {
1320
- id?: string;
1321
- scenarioId?: string;
1322
- splitTag?: RunSplitTag;
1323
- runtimeEvents?: unknown;
1324
- [key: string]: unknown;
1325
- }
1326
- interface RuntimeTrajectoryRunRecord {
1327
- runId: string;
1328
- scenarioId?: string;
1329
- splitTag: RunSplitTag;
1330
- }
1331
- interface RuntimeTrajectoryEvidenceSummary {
1332
- recordCount: number;
1333
- recordWithRuntimeEventsCount: number;
1334
- runtimeRunCount: number;
1335
- lifecycleEventCount: number;
1336
- defaultedSplitCount: number;
1337
- }
1338
- interface RuntimeTrajectoryEvidenceProjection {
1339
- runs: RuntimeTrajectoryRunRecord[];
1340
- events: RuntimeTrajectoryHookEvent[];
1341
- summary: RuntimeTrajectoryEvidenceSummary;
1342
- diagnostics: string[];
1343
- }
1344
- interface ProjectRuntimeTrajectoryEvidenceOptions<TRecord extends RuntimeTrajectoryRecord = RuntimeTrajectoryRecord> {
1345
- records: TRecord[];
1346
- defaultSplitTag?: RunSplitTag;
1347
- recordIdOf?: (record: TRecord, index: number) => string | undefined;
1348
- scenarioIdOf?: (record: TRecord, index: number) => string | undefined;
1349
- }
1350
-
591
+ //#endregion
592
+ //#region src/belief-state/runtime-benchmark-corpus.d.ts
1351
593
  type RuntimeBenchmarkTrajectoryRecord = RuntimeTrajectoryRecord & {
1352
- benchmark?: unknown;
1353
- condition?: unknown;
1354
- instanceId?: unknown;
1355
- runtimeDecisionPoints?: unknown;
594
+ benchmark?: unknown;
595
+ condition?: unknown;
596
+ instanceId?: unknown;
597
+ runtimeDecisionPoints?: unknown;
1356
598
  };
1357
599
  interface BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions extends Omit<BuildRuntimeBeliefPhase0MeasurementOptions, 'runs' | 'events' | 'decisions' | 'labels'> {
1358
- records: RuntimeBenchmarkTrajectoryRecord[];
1359
- decisions?: RuntimeBeliefDecisionPoint[];
1360
- defaultSplitTag?: ProjectRuntimeTrajectoryEvidenceOptions['defaultSplitTag'];
1361
- labels?: RuntimeBeliefDecisionLabel[];
600
+ records: RuntimeBenchmarkTrajectoryRecord[];
601
+ decisions?: RuntimeBeliefDecisionPoint[];
602
+ defaultSplitTag?: ProjectRuntimeTrajectoryEvidenceOptions['defaultSplitTag'];
603
+ labels?: RuntimeBeliefDecisionLabel[];
1362
604
  }
1363
605
  interface RuntimeBenchmarkBeliefPhase0Summary {
1364
- decisionCount: number;
1365
- labelCount: number;
606
+ decisionCount: number;
607
+ labelCount: number;
1366
608
  }
1367
609
  interface RuntimeBenchmarkBeliefPhase0Measurement {
1368
- runs: RuntimeBeliefPhase0RunRecord[];
1369
- events: RuntimeBeliefHookEvent[];
1370
- decisions: RuntimeBeliefDecisionPoint[];
1371
- labels: RuntimeBeliefDecisionLabel[];
1372
- trajectory: RuntimeTrajectoryEvidenceProjection;
1373
- measurement: RuntimeBeliefPhase0Measurement;
1374
- summary: RuntimeBenchmarkBeliefPhase0Summary;
1375
- diagnostics: string[];
610
+ runs: RuntimeBeliefPhase0RunRecord[];
611
+ events: RuntimeBeliefHookEvent[];
612
+ decisions: RuntimeBeliefDecisionPoint[];
613
+ labels: RuntimeBeliefDecisionLabel[];
614
+ trajectory: RuntimeTrajectoryEvidenceProjection;
615
+ measurement: RuntimeBeliefPhase0Measurement;
616
+ summary: RuntimeBenchmarkBeliefPhase0Summary;
617
+ diagnostics: string[];
1376
618
  }
1377
619
  declare function buildRuntimeBenchmarkBeliefPhase0Measurement(options: BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions): RuntimeBenchmarkBeliefPhase0Measurement;
1378
-
1379
- export { type AnalyzeBeliefDecisionCorpusOptions, type AnalyzeBeliefPolicyOpeOptions, type AnalyzeBeliefPolicyOptions, BELIEF_DECISION_KINDS, BELIEF_EVALUATION_CRITERIA, BELIEF_EVIDENCE_QUALITIES, BELIEF_EVIDENCE_SOURCES, type BeliefCalibrationOptions, type BeliefCalibrationRegion, type BeliefCalibrationStatus, type BeliefDecisionCorpusEvaluation, type BeliefDecisionExtractionDiagnostic, type BeliefDecisionExtractionReport, type BeliefDecisionInventoryBucket, type BeliefDecisionInventoryReport, type BeliefDecisionKind, type BeliefDecisionOutcome, type BeliefDecisionPoint, type BeliefDecisionReason, type BeliefDecisionReasonCode, type BeliefDecisionResearchEvidencePacket, type BeliefDecisionTargetSelection, type BeliefEvaluationCriterionId, type BeliefEvaluationStatus, type BeliefEvidenceQuality, type BeliefEvidenceRef, type BeliefEvidenceSource, type BeliefOffPolicyTrajectoryReport, type BeliefOpeOptions, type BeliefOpeReport, type BeliefOpeStatus, type BeliefOpeSupportDiagnostics, type BeliefOpeTargetPolicy, type BeliefPolicyAction, type BeliefPolicyDecision, type BeliefPolicyEvaluationReport, type BeliefResearchClaimScope, type BeliefResearchEvidenceGate, type BeliefResearchEvidenceStatus, type BeliefResearchGateId, type BeliefRuntimeHookCollector, type BeliefSelectivePolicy, type BeliefSelectivePolicyMetrics, type BeliefShadowProbeDiagnostic, type BeliefShadowProbeEvidenceRef, type BeliefShadowProbeInput, type BeliefShadowProbeRecord, type BeliefShadowProbeResponse, type BeliefShadowProbeRun, type BeliefShadowProbeSummary, type BeliefUtilityOptions, type BuildBeliefDecisionResearchEvidencePacketOptions, type BuildCodeAgentBeliefEvidenceCorpusOptions, type BuildRuntimeBeliefPhase0MeasurementOptions, type BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions, type CodeAgentBeliefDecisionTargetId, type CodeAgentBeliefEvidenceCorpus, type CodeAgentBeliefSession, type EvaluateBeliefSelectivePolicyOptions, type ExtractBeliefDecisionPointsOptions, type ExtractCodeAgentBeliefDecisionPointsOptions, type RunBeliefShadowProbeOptions, type RuntimeBeliefConversionDiagnostic, type RuntimeBeliefDecisionEvidenceRef, type RuntimeBeliefDecisionLabel, type RuntimeBeliefDecisionPoint, type RuntimeBeliefDecisionPointOptions, type RuntimeBeliefDecisionPointReport, type RuntimeBeliefHookContext, type RuntimeBeliefHookEvent, type RuntimeBeliefHooks, type RuntimeBeliefPhase0Measurement, type RuntimeBeliefPhase0MeasurementSummary, type RuntimeBeliefPhase0RunRecord, type RuntimeBeliefShadowProbeInputOptions, type RuntimeBeliefShadowProbeInputReport, type RuntimeBenchmarkBeliefPhase0Measurement, type RuntimeBenchmarkBeliefPhase0Summary, type SelectBeliefDecisionTargetOptions, analyzeBeliefDecisionCorpus, analyzeBeliefPolicy, beliefDecisionsToOffPolicyTrajectories, buildBeliefDecisionResearchEvidencePacket, buildCodeAgentBeliefEvidenceCorpus, buildRuntimeBeliefPhase0Measurement, buildRuntimeBenchmarkBeliefPhase0Measurement, calibrateBeliefDecisions, createBeliefRuntimeHookCollector, embeddedBeliefOpeTargetPolicy, evaluateBeliefOffPolicy, evaluateBeliefSelectivePolicy, extractBeliefDecisionPoints, extractCodeAgentBeliefDecisionPoints, formatBeliefShadowProbePrompt, inventoryBeliefDecisionPoints, isBeliefDecisionKind, isBeliefEvidenceSource, runBeliefShadowProbe, runtimeDecisionPointToBeliefDecisionPoint, runtimeDecisionPointToBeliefShadowProbeInput, selectBeliefDecisionTarget, thresholdSelectivePolicy };
620
+ //#endregion
621
+ export { AnalyzeBeliefDecisionCorpusOptions, AnalyzeBeliefPolicyOpeOptions, AnalyzeBeliefPolicyOptions, BELIEF_DECISION_KINDS, BELIEF_EVALUATION_CRITERIA, BELIEF_EVIDENCE_QUALITIES, BELIEF_EVIDENCE_SOURCES, BeliefCalibrationOptions, BeliefCalibrationRegion, BeliefCalibrationStatus, BeliefDecisionCorpusEvaluation, BeliefDecisionExtractionDiagnostic, BeliefDecisionExtractionReport, BeliefDecisionInventoryBucket, BeliefDecisionInventoryReport, BeliefDecisionKind, BeliefDecisionOutcome, BeliefDecisionPoint, BeliefDecisionReason, BeliefDecisionReasonCode, BeliefDecisionResearchEvidencePacket, BeliefDecisionTargetSelection, BeliefEvaluationCriterionId, BeliefEvaluationStatus, BeliefEvidenceQuality, BeliefEvidenceRef, BeliefEvidenceSource, BeliefOffPolicyTrajectoryReport, BeliefOpeOptions, BeliefOpeReport, BeliefOpeStatus, BeliefOpeSupportDiagnostics, BeliefOpeTargetPolicy, BeliefPolicyAction, BeliefPolicyDecision, BeliefPolicyEvaluationReport, BeliefResearchClaimScope, BeliefResearchEvidenceGate, BeliefResearchEvidenceStatus, BeliefResearchGateId, BeliefRuntimeHookCollector, BeliefSelectivePolicy, BeliefSelectivePolicyMetrics, BeliefShadowProbeDiagnostic, BeliefShadowProbeEvidenceRef, BeliefShadowProbeInput, BeliefShadowProbeRecord, BeliefShadowProbeResponse, BeliefShadowProbeRun, BeliefShadowProbeSummary, BeliefUtilityOptions, BuildBeliefDecisionResearchEvidencePacketOptions, BuildCodeAgentBeliefEvidenceCorpusOptions, BuildRuntimeBeliefPhase0MeasurementOptions, BuildRuntimeBenchmarkBeliefPhase0MeasurementOptions, CodeAgentBeliefDecisionTargetId, CodeAgentBeliefEvidenceCorpus, CodeAgentBeliefSession, EvaluateBeliefSelectivePolicyOptions, ExtractBeliefDecisionPointsOptions, ExtractCodeAgentBeliefDecisionPointsOptions, RunBeliefShadowProbeOptions, RuntimeBeliefConversionDiagnostic, RuntimeBeliefDecisionEvidenceRef, RuntimeBeliefDecisionLabel, RuntimeBeliefDecisionPoint, RuntimeBeliefDecisionPointOptions, RuntimeBeliefDecisionPointReport, RuntimeBeliefHookContext, RuntimeBeliefHookEvent, RuntimeBeliefHooks, RuntimeBeliefPhase0Measurement, RuntimeBeliefPhase0MeasurementSummary, RuntimeBeliefPhase0RunRecord, RuntimeBeliefShadowProbeInputOptions, RuntimeBeliefShadowProbeInputReport, RuntimeBenchmarkBeliefPhase0Measurement, RuntimeBenchmarkBeliefPhase0Summary, SelectBeliefDecisionTargetOptions, analyzeBeliefDecisionCorpus, analyzeBeliefPolicy, beliefDecisionsToOffPolicyTrajectories, buildBeliefDecisionResearchEvidencePacket, buildCodeAgentBeliefEvidenceCorpus, buildRuntimeBeliefPhase0Measurement, buildRuntimeBenchmarkBeliefPhase0Measurement, calibrateBeliefDecisions, createBeliefRuntimeHookCollector, embeddedBeliefOpeTargetPolicy, evaluateBeliefOffPolicy, evaluateBeliefSelectivePolicy, extractBeliefDecisionPoints, extractCodeAgentBeliefDecisionPoints, formatBeliefShadowProbePrompt, inventoryBeliefDecisionPoints, isBeliefDecisionKind, isBeliefEvidenceSource, runBeliefShadowProbe, runtimeDecisionPointToBeliefDecisionPoint, runtimeDecisionPointToBeliefShadowProbeInput, selectBeliefDecisionTarget, thresholdSelectivePolicy };
622
+ //# sourceMappingURL=index.d.ts.map