@tangle-network/agent-eval 0.129.0 → 0.130.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (428) hide show
  1. package/CHANGELOG.md +21 -0
  2. package/README.md +2 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-BJgDGkAD.js +754 -0
  36. package/dist/benchmarks-BJgDGkAD.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-aKJt6emI.js +3886 -0
  46. package/dist/campaign-aKJt6emI.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CD_WZ_Xr.d.ts +2250 -0
  116. package/dist/index-CD_WZ_Xr.d.ts.map +1 -0
  117. package/dist/index-DSC51roc.d.ts +102 -0
  118. package/dist/index-DSC51roc.d.ts.map +1 -0
  119. package/dist/index-Em67JBjs.d.ts +335 -0
  120. package/dist/index-Em67JBjs.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-CUmHkGbI.js +7718 -0
  253. package/dist/skillopt-optimization-method-CUmHkGbI.js.map +1 -0
  254. package/dist/skillopt-optimization-method-CWKVTnks.d.ts +1740 -0
  255. package/dist/skillopt-optimization-method-CWKVTnks.d.ts.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/campaign-proposers.md +1 -0
  301. package/package.json +17 -9
  302. package/dist/benchmarks/index.js.map +0 -1
  303. package/dist/campaign/index.js.map +0 -1
  304. package/dist/chunk-2QU3YOPR.js +0 -7374
  305. package/dist/chunk-2QU3YOPR.js.map +0 -1
  306. package/dist/chunk-3OCR4R5I.js +0 -728
  307. package/dist/chunk-3OCR4R5I.js.map +0 -1
  308. package/dist/chunk-3RF76KTD.js +0 -84
  309. package/dist/chunk-3RF76KTD.js.map +0 -1
  310. package/dist/chunk-56TAVBOK.js +0 -698
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7FO3TNPI.js +0 -232
  314. package/dist/chunk-7FO3TNPI.js.map +0 -1
  315. package/dist/chunk-7ZZMD7UK.js +0 -386
  316. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  317. package/dist/chunk-BOD4O7OF.js +0 -40
  318. package/dist/chunk-BOD4O7OF.js.map +0 -1
  319. package/dist/chunk-BSO5JDQH.js +0 -2335
  320. package/dist/chunk-BSO5JDQH.js.map +0 -1
  321. package/dist/chunk-C6LXANRU.js +0 -1550
  322. package/dist/chunk-C6LXANRU.js.map +0 -1
  323. package/dist/chunk-DODXQREJ.js +0 -752
  324. package/dist/chunk-DODXQREJ.js.map +0 -1
  325. package/dist/chunk-DRYIUNWY.js +0 -622
  326. package/dist/chunk-DRYIUNWY.js.map +0 -1
  327. package/dist/chunk-E7QXT7SX.js +0 -183
  328. package/dist/chunk-E7QXT7SX.js.map +0 -1
  329. package/dist/chunk-EG66UGL4.js +0 -341
  330. package/dist/chunk-EG66UGL4.js.map +0 -1
  331. package/dist/chunk-FXTVJPYD.js +0 -576
  332. package/dist/chunk-FXTVJPYD.js.map +0 -1
  333. package/dist/chunk-G7MGMCZD.js +0 -153
  334. package/dist/chunk-G7MGMCZD.js.map +0 -1
  335. package/dist/chunk-GGE4NNQT.js +0 -65
  336. package/dist/chunk-GGE4NNQT.js.map +0 -1
  337. package/dist/chunk-H23X7XKK.js +0 -181
  338. package/dist/chunk-H23X7XKK.js.map +0 -1
  339. package/dist/chunk-HHWE3POT.js +0 -94
  340. package/dist/chunk-HHWE3POT.js.map +0 -1
  341. package/dist/chunk-HPWUNB47.js +0 -289
  342. package/dist/chunk-HPWUNB47.js.map +0 -1
  343. package/dist/chunk-IYCLP2N2.js +0 -766
  344. package/dist/chunk-IYCLP2N2.js.map +0 -1
  345. package/dist/chunk-JHCHEVET.js +0 -274
  346. package/dist/chunk-JHCHEVET.js.map +0 -1
  347. package/dist/chunk-JQSF5DQT.js +0 -701
  348. package/dist/chunk-JQSF5DQT.js.map +0 -1
  349. package/dist/chunk-K4DBDHLK.js +0 -158
  350. package/dist/chunk-K4DBDHLK.js.map +0 -1
  351. package/dist/chunk-K6N6XJJX.js +0 -306
  352. package/dist/chunk-K6N6XJJX.js.map +0 -1
  353. package/dist/chunk-M4YBQKIJ.js +0 -1040
  354. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  355. package/dist/chunk-MA6HLL3S.js +0 -65
  356. package/dist/chunk-MA6HLL3S.js.map +0 -1
  357. package/dist/chunk-MAZ26DC7.js +0 -99
  358. package/dist/chunk-MAZ26DC7.js.map +0 -1
  359. package/dist/chunk-NPCTHQIO.js +0 -91
  360. package/dist/chunk-NPCTHQIO.js.map +0 -1
  361. package/dist/chunk-NY44NC4A.js +0 -1056
  362. package/dist/chunk-NY44NC4A.js.map +0 -1
  363. package/dist/chunk-OIUOT4QD.js +0 -44
  364. package/dist/chunk-OIUOT4QD.js.map +0 -1
  365. package/dist/chunk-ONWEPEDO.js +0 -57
  366. package/dist/chunk-ONWEPEDO.js.map +0 -1
  367. package/dist/chunk-OWN5NPMC.js +0 -152
  368. package/dist/chunk-OWN5NPMC.js.map +0 -1
  369. package/dist/chunk-P6FYH6K4.js +0 -1161
  370. package/dist/chunk-P6FYH6K4.js.map +0 -1
  371. package/dist/chunk-PC4UYEBM.js +0 -166
  372. package/dist/chunk-PC4UYEBM.js.map +0 -1
  373. package/dist/chunk-PC5DOSM7.js +0 -579
  374. package/dist/chunk-PC5DOSM7.js.map +0 -1
  375. package/dist/chunk-PXE2VKMX.js +0 -140
  376. package/dist/chunk-PXE2VKMX.js.map +0 -1
  377. package/dist/chunk-PZ5AY32C.js +0 -10
  378. package/dist/chunk-PZ5AY32C.js.map +0 -1
  379. package/dist/chunk-QB6BDBP2.js +0 -4464
  380. package/dist/chunk-QB6BDBP2.js.map +0 -1
  381. package/dist/chunk-RXHCETDZ.js +0 -536
  382. package/dist/chunk-RXHCETDZ.js.map +0 -1
  383. package/dist/chunk-RZTMDUO7.js +0 -49
  384. package/dist/chunk-RZTMDUO7.js.map +0 -1
  385. package/dist/chunk-SFLLL76A.js +0 -669
  386. package/dist/chunk-SFLLL76A.js.map +0 -1
  387. package/dist/chunk-SZLVEKMJ.js +0 -1446
  388. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  389. package/dist/chunk-T4SQEITX.js +0 -95
  390. package/dist/chunk-T4SQEITX.js.map +0 -1
  391. package/dist/chunk-T6RLYGAD.js +0 -158
  392. package/dist/chunk-T6RLYGAD.js.map +0 -1
  393. package/dist/chunk-TJVT4QFF.js +0 -911
  394. package/dist/chunk-TJVT4QFF.js.map +0 -1
  395. package/dist/chunk-TQ7LNKZ3.js +0 -136
  396. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  397. package/dist/chunk-U4L7JRPZ.js +0 -1706
  398. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  399. package/dist/chunk-U4PHLT2N.js +0 -419
  400. package/dist/chunk-U4PHLT2N.js.map +0 -1
  401. package/dist/chunk-VCZ5FQYW.js +0 -928
  402. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  403. package/dist/chunk-VI2UW6B6.js +0 -162
  404. package/dist/chunk-VI2UW6B6.js.map +0 -1
  405. package/dist/chunk-VQMK5FMP.js +0 -247
  406. package/dist/chunk-VQMK5FMP.js.map +0 -1
  407. package/dist/chunk-WGXIEX7P.js +0 -116
  408. package/dist/chunk-WGXIEX7P.js.map +0 -1
  409. package/dist/chunk-WVATSFCP.js +0 -1553
  410. package/dist/chunk-WVATSFCP.js.map +0 -1
  411. package/dist/chunk-X4YIBDER.js +0 -1662
  412. package/dist/chunk-X4YIBDER.js.map +0 -1
  413. package/dist/chunk-YQN4ICPP.js +0 -355
  414. package/dist/chunk-YQN4ICPP.js.map +0 -1
  415. package/dist/chunk-ZET2UAYW.js +0 -89
  416. package/dist/chunk-ZET2UAYW.js.map +0 -1
  417. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  418. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  419. package/dist/control.js.map +0 -1
  420. package/dist/hosted/index.js.map +0 -1
  421. package/dist/matrix/index.js.map +0 -1
  422. package/dist/reporting.js.map +0 -1
  423. package/dist/rollout/index.js.map +0 -1
  424. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  425. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  426. package/dist/supervisor-run/index.js.map +0 -1
  427. package/dist/traces.js.map +0 -1
  428. package/dist/wire/index.js.map +0 -1
@@ -1,1911 +1,1819 @@
1
- import {
2
- fromClaudeCodeSession,
3
- fromCodexSession,
4
- fromKimiCodeSession,
5
- fromOpenCodeSession,
6
- fromPiSession,
7
- observeCodeAgentSession
8
- } from "../chunk-SZLVEKMJ.js";
9
- import {
10
- calibrationFromPairs
11
- } from "../chunk-NPCTHQIO.js";
12
- import {
13
- projectRuntimeTrajectoryEvidence
14
- } from "../chunk-T4SQEITX.js";
15
- import {
16
- offPolicyEstimateAll
17
- } from "../chunk-T6RLYGAD.js";
18
- import {
19
- confidenceInterval
20
- } from "../chunk-ZHTZ4EYI.js";
21
- import "../chunk-VI2UW6B6.js";
22
- import "../chunk-PXE2VKMX.js";
23
- import {
24
- trainingScore
25
- } from "../chunk-OIUOT4QD.js";
26
- import {
27
- ValidationError
28
- } from "../chunk-ONWEPEDO.js";
29
- import "../chunk-PZ5AY32C.js";
30
-
31
- // src/belief-state/calibration.ts
1
+ import { s as ValidationError } from "../errors-8YnH8WlF.js";
2
+ import { a as confidenceInterval } from "../statistics-CnnxdpOg.js";
3
+ import { s as trainingScore } from "../reward-nw2xZGZG.js";
4
+ import { n as projectRuntimeTrajectoryEvidence } from "../runtime-trajectory-1gyaTOoC.js";
5
+ import { r as offPolicyEstimateAll } from "../off-policy-DvgzvtIx.js";
6
+ import { n as calibrationFromPairs } from "../calibration-CNWWA6K8.js";
7
+ import { a as fromPiSession, c as observeCodeAgentSession, i as fromOpenCodeSession, n as fromCodexSession, r as fromKimiCodeSession, t as fromClaudeCodeSession } from "../code-agent-session-BjkMTQ7H.js";
8
+ //#region src/belief-state/calibration.ts
32
9
  function calibrateBeliefDecisions(points, options = {}) {
33
- const filtered = filterCalibrationRegion(points, options);
34
- const pairs = filtered.filter((point) => typeof point.confidence === "number" && point.outcome).map((point) => ({
35
- evalScore: point.confidence,
36
- outcome: outcomeScore(point)
37
- })).filter((pair) => Number.isFinite(pair.outcome));
38
- const minPairs = options.minPairs ?? 10;
39
- if (pairs.length < minPairs) return null;
40
- return calibrationFromPairs(pairs, "belief-confidence", "decision-outcome", {
41
- bins: options.bins ?? 5,
42
- range: { lo: 0, hi: 1 }
43
- });
10
+ const pairs = filterCalibrationRegion(points, options).filter((point) => typeof point.confidence === "number" && point.outcome).map((point) => ({
11
+ evalScore: point.confidence,
12
+ outcome: outcomeScore$1(point)
13
+ })).filter((pair) => Number.isFinite(pair.outcome));
14
+ const minPairs = options.minPairs ?? 10;
15
+ if (pairs.length < minPairs) return null;
16
+ return calibrationFromPairs(pairs, "belief-confidence", "decision-outcome", {
17
+ bins: options.bins ?? 5,
18
+ range: {
19
+ lo: 0,
20
+ hi: 1
21
+ }
22
+ });
44
23
  }
45
24
  function filterCalibrationRegion(points, options) {
46
- const region = options.region ?? "all";
47
- if (region === "all") return points;
48
- const policy = options.policy;
49
- if (!policy) {
50
- throw new ValidationError(
51
- `calibrateBeliefDecisions: policy is required when region is "${region}"`
52
- );
53
- }
54
- return points.filter((point) => {
55
- const accepted = policy.decide(point).action === "accept";
56
- return region === "accepted" ? accepted : !accepted;
57
- });
58
- }
59
- function outcomeScore(point) {
60
- if (typeof point.outcome?.reward === "number") return point.outcome.reward;
61
- if (typeof point.outcome?.score === "number") return point.outcome.score;
62
- if (point.outcome?.success === true) return 1;
63
- if (point.outcome?.success === false) return 0;
64
- return Number.NaN;
65
- }
66
-
67
- // src/belief-state/ope.ts
25
+ const region = options.region ?? "all";
26
+ if (region === "all") return points;
27
+ const policy = options.policy;
28
+ if (!policy) throw new ValidationError(`calibrateBeliefDecisions: policy is required when region is "${region}"`);
29
+ return points.filter((point) => {
30
+ const accepted = policy.decide(point).action === "accept";
31
+ return region === "accepted" ? accepted : !accepted;
32
+ });
33
+ }
34
+ function outcomeScore$1(point) {
35
+ if (typeof point.outcome?.reward === "number") return point.outcome.reward;
36
+ if (typeof point.outcome?.score === "number") return point.outcome.score;
37
+ if (point.outcome?.success === true) return 1;
38
+ if (point.outcome?.success === false) return 0;
39
+ return NaN;
40
+ }
41
+ //#endregion
42
+ //#region src/belief-state/ope.ts
68
43
  function embeddedBeliefOpeTargetPolicy(id = "embedded-target-prob") {
69
- return {
70
- id,
71
- targetProbOf(point) {
72
- return point.targetProb;
73
- },
74
- qHatChosenOf(point) {
75
- return point.qHatChosen;
76
- },
77
- vHatTargetOf(point) {
78
- return point.vHatTarget;
79
- }
80
- };
44
+ return {
45
+ id,
46
+ targetProbOf(point) {
47
+ return point.targetProb;
48
+ },
49
+ qHatChosenOf(point) {
50
+ return point.qHatChosen;
51
+ },
52
+ vHatTargetOf(point) {
53
+ return point.vHatTarget;
54
+ }
55
+ };
81
56
  }
82
57
  function beliefDecisionsToOffPolicyTrajectories(points, targetPolicy, options = {}) {
83
- const trajectories = [];
84
- const diagnostics = [];
85
- for (const point of points) {
86
- if (!point.outcome) {
87
- diagnostics.push(`${point.id}: missing outcome`);
88
- continue;
89
- }
90
- if (!isBehaviorProbability(point.behaviorProb)) {
91
- diagnostics.push(`${point.id}: invalid behaviorProb ${formatProbability(point.behaviorProb)}`);
92
- continue;
93
- }
94
- let targetProb;
95
- let qHatChosen;
96
- let vHatTarget;
97
- try {
98
- targetProb = targetPolicy.targetProbOf(point);
99
- qHatChosen = targetPolicy.qHatChosenOf?.(point);
100
- vHatTarget = targetPolicy.vHatTargetOf?.(point);
101
- } catch (error) {
102
- diagnostics.push(
103
- `${point.id}: target policy ${targetPolicy.id} threw (${errorMessage(error)})`
104
- );
105
- continue;
106
- }
107
- if (!isTargetProbability(targetProb)) {
108
- diagnostics.push(`${point.id}: invalid targetProb ${formatProbability(targetProb)}`);
109
- continue;
110
- }
111
- const hasQHatChosen = qHatChosen !== null && qHatChosen !== void 0;
112
- const hasVHatTarget = vHatTarget !== null && vHatTarget !== void 0;
113
- if (hasQHatChosen !== hasVHatTarget) {
114
- diagnostics.push(`${point.id}: qHatChosen and vHatTarget must be supplied together`);
115
- continue;
116
- }
117
- if (hasQHatChosen && hasVHatTarget && (!isTargetProbability(qHatChosen) || !isTargetProbability(vHatTarget))) {
118
- diagnostics.push(
119
- `${point.id}: invalid contextual Q pair qHatChosen=${formatProbability(qHatChosen)} vHatTarget=${formatProbability(vHatTarget)}`
120
- );
121
- continue;
122
- }
123
- trajectories.push({
124
- runId: point.id,
125
- reward: rewardOf(point),
126
- behaviorProb: point.behaviorProb,
127
- targetProb,
128
- ...qHatChosen !== void 0 ? { qHatChosen } : {},
129
- ...vHatTarget !== void 0 ? { vHatTarget } : {}
130
- });
131
- }
132
- return {
133
- targetPolicyId: targetPolicy.id,
134
- trajectories,
135
- dropped: points.length - trajectories.length,
136
- diagnostics: compactDiagnostics(diagnostics, options.maxDiagnostics ?? 20)
137
- };
58
+ const trajectories = [];
59
+ const diagnostics = [];
60
+ for (const point of points) {
61
+ if (!point.outcome) {
62
+ diagnostics.push(`${point.id}: missing outcome`);
63
+ continue;
64
+ }
65
+ if (!isBehaviorProbability(point.behaviorProb)) {
66
+ diagnostics.push(`${point.id}: invalid behaviorProb ${formatProbability(point.behaviorProb)}`);
67
+ continue;
68
+ }
69
+ let targetProb;
70
+ let qHatChosen;
71
+ let vHatTarget;
72
+ try {
73
+ targetProb = targetPolicy.targetProbOf(point);
74
+ qHatChosen = targetPolicy.qHatChosenOf?.(point);
75
+ vHatTarget = targetPolicy.vHatTargetOf?.(point);
76
+ } catch (error) {
77
+ diagnostics.push(`${point.id}: target policy ${targetPolicy.id} threw (${errorMessage$1(error)})`);
78
+ continue;
79
+ }
80
+ if (!isTargetProbability(targetProb)) {
81
+ diagnostics.push(`${point.id}: invalid targetProb ${formatProbability(targetProb)}`);
82
+ continue;
83
+ }
84
+ const hasQHatChosen = qHatChosen !== null && qHatChosen !== void 0;
85
+ const hasVHatTarget = vHatTarget !== null && vHatTarget !== void 0;
86
+ if (hasQHatChosen !== hasVHatTarget) {
87
+ diagnostics.push(`${point.id}: qHatChosen and vHatTarget must be supplied together`);
88
+ continue;
89
+ }
90
+ if (hasQHatChosen && hasVHatTarget && (!isTargetProbability(qHatChosen) || !isTargetProbability(vHatTarget))) {
91
+ diagnostics.push(`${point.id}: invalid contextual Q pair qHatChosen=${formatProbability(qHatChosen)} vHatTarget=${formatProbability(vHatTarget)}`);
92
+ continue;
93
+ }
94
+ trajectories.push({
95
+ runId: point.id,
96
+ reward: rewardOf$1(point),
97
+ behaviorProb: point.behaviorProb,
98
+ targetProb,
99
+ ...qHatChosen !== void 0 ? { qHatChosen } : {},
100
+ ...vHatTarget !== void 0 ? { vHatTarget } : {}
101
+ });
102
+ }
103
+ return {
104
+ targetPolicyId: targetPolicy.id,
105
+ trajectories,
106
+ dropped: points.length - trajectories.length,
107
+ diagnostics: compactDiagnostics(diagnostics, options.maxDiagnostics ?? 20)
108
+ };
138
109
  }
139
110
  function evaluateBeliefOffPolicy(points, targetPolicy, options = {}) {
140
- const trajectoryReport = beliefDecisionsToOffPolicyTrajectories(points, targetPolicy, options);
141
- const { trajectories } = trajectoryReport;
142
- const estimates = offPolicyEstimateAll(trajectories, options);
143
- const support = supportDiagnostics(estimates.dr, {
144
- minEffectiveSampleSize: options.minEffectiveSampleSize ?? 30,
145
- minEffectiveSampleRatio: options.minEffectiveSampleRatio ?? 0.25,
146
- dropped: trajectoryReport.dropped,
147
- diagnostics: trajectoryReport.diagnostics
148
- });
149
- return { targetPolicyId: targetPolicy.id, ...estimates, support };
111
+ const trajectoryReport = beliefDecisionsToOffPolicyTrajectories(points, targetPolicy, options);
112
+ const { trajectories } = trajectoryReport;
113
+ const estimates = offPolicyEstimateAll(trajectories, options);
114
+ const support = supportDiagnostics(estimates.dr, {
115
+ minEffectiveSampleSize: options.minEffectiveSampleSize ?? 30,
116
+ minEffectiveSampleRatio: options.minEffectiveSampleRatio ?? .25,
117
+ dropped: trajectoryReport.dropped,
118
+ diagnostics: trajectoryReport.diagnostics
119
+ });
120
+ return {
121
+ targetPolicyId: targetPolicy.id,
122
+ ...estimates,
123
+ support
124
+ };
150
125
  }
151
126
  function supportDiagnostics(estimate, options) {
152
- const ratio2 = estimate.n > 0 ? estimate.effectiveSampleSize / estimate.n : 0;
153
- const reasons = [...options.diagnostics];
154
- if (estimate.n === 0) {
155
- reasons.push("no valid OPE trajectories");
156
- }
157
- if (options.dropped > 0) {
158
- reasons.push(`dropped ${options.dropped} unsupported decision(s)`);
159
- }
160
- if (estimate.effectiveSampleSize < options.minEffectiveSampleSize) {
161
- reasons.push(
162
- `effective sample size ${estimate.effectiveSampleSize.toFixed(2)} below ${options.minEffectiveSampleSize}`
163
- );
164
- }
165
- if (ratio2 < options.minEffectiveSampleRatio) {
166
- reasons.push(
167
- `effective sample ratio ${ratio2.toFixed(2)} below ${options.minEffectiveSampleRatio}`
168
- );
169
- }
170
- if (estimate.maxImportanceWeight > 10) {
171
- reasons.push(`max importance weight ${estimate.maxImportanceWeight.toFixed(2)} is high`);
172
- }
173
- return {
174
- supported: reasons.length === 0,
175
- n: estimate.n,
176
- dropped: options.dropped,
177
- effectiveSampleSize: estimate.effectiveSampleSize,
178
- effectiveSampleRatio: ratio2,
179
- maxImportanceWeight: estimate.maxImportanceWeight,
180
- reasons
181
- };
182
- }
183
- function rewardOf(point) {
184
- if (typeof point.outcome?.reward === "number") return point.outcome.reward;
185
- if (typeof point.outcome?.score === "number") return point.outcome.score;
186
- if (point.outcome?.success === true) return 1;
187
- return 0;
127
+ const ratio = estimate.n > 0 ? estimate.effectiveSampleSize / estimate.n : 0;
128
+ const reasons = [...options.diagnostics];
129
+ if (estimate.n === 0) reasons.push("no valid OPE trajectories");
130
+ if (options.dropped > 0) reasons.push(`dropped ${options.dropped} unsupported decision(s)`);
131
+ if (estimate.effectiveSampleSize < options.minEffectiveSampleSize) reasons.push(`effective sample size ${estimate.effectiveSampleSize.toFixed(2)} below ${options.minEffectiveSampleSize}`);
132
+ if (ratio < options.minEffectiveSampleRatio) reasons.push(`effective sample ratio ${ratio.toFixed(2)} below ${options.minEffectiveSampleRatio}`);
133
+ if (estimate.maxImportanceWeight > 10) reasons.push(`max importance weight ${estimate.maxImportanceWeight.toFixed(2)} is high`);
134
+ return {
135
+ supported: reasons.length === 0,
136
+ n: estimate.n,
137
+ dropped: options.dropped,
138
+ effectiveSampleSize: estimate.effectiveSampleSize,
139
+ effectiveSampleRatio: ratio,
140
+ maxImportanceWeight: estimate.maxImportanceWeight,
141
+ reasons
142
+ };
143
+ }
144
+ function rewardOf$1(point) {
145
+ if (typeof point.outcome?.reward === "number") return point.outcome.reward;
146
+ if (typeof point.outcome?.score === "number") return point.outcome.score;
147
+ if (point.outcome?.success === true) return 1;
148
+ return 0;
188
149
  }
189
150
  function isBehaviorProbability(value) {
190
- return typeof value === "number" && Number.isFinite(value) && value > 0 && value <= 1;
151
+ return typeof value === "number" && Number.isFinite(value) && value > 0 && value <= 1;
191
152
  }
192
153
  function isTargetProbability(value) {
193
- return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
154
+ return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
194
155
  }
195
156
  function formatProbability(value) {
196
- return typeof value === "number" ? String(value) : String(value ?? "missing");
157
+ return typeof value === "number" ? String(value) : String(value ?? "missing");
197
158
  }
198
- function errorMessage(error) {
199
- return error instanceof Error ? error.message : String(error);
159
+ function errorMessage$1(error) {
160
+ return error instanceof Error ? error.message : String(error);
200
161
  }
201
162
  function compactDiagnostics(diagnostics, maxDiagnostics) {
202
- if (diagnostics.length <= maxDiagnostics) return diagnostics;
203
- return [
204
- ...diagnostics.slice(0, maxDiagnostics),
205
- `${diagnostics.length - maxDiagnostics} additional OPE diagnostic(s) omitted`
206
- ];
207
- }
208
-
209
- // src/belief-state/selective.ts
210
- var DEFAULT_UTILITY = {
211
- successUtility: 1,
212
- failureUtility: -1,
213
- deferUtility: 0,
214
- verifyCost: 0.05,
215
- askCost: 0.05,
216
- retryCost: 0.1,
217
- stopUtility: 0,
218
- costWeight: 1
163
+ if (diagnostics.length <= maxDiagnostics) return diagnostics;
164
+ return [...diagnostics.slice(0, maxDiagnostics), `${diagnostics.length - maxDiagnostics} additional OPE diagnostic(s) omitted`];
165
+ }
166
+ //#endregion
167
+ //#region src/belief-state/selective.ts
168
+ const DEFAULT_UTILITY = {
169
+ successUtility: 1,
170
+ failureUtility: -1,
171
+ deferUtility: 0,
172
+ verifyCost: .05,
173
+ askCost: .05,
174
+ retryCost: .1,
175
+ stopUtility: 0,
176
+ costWeight: 1
219
177
  };
220
178
  function thresholdSelectivePolicy(options) {
221
- const threshold = options.confidenceThreshold;
222
- if (!Number.isFinite(threshold) || threshold < 0 || threshold > 1) {
223
- throw new ValidationError(
224
- `thresholdSelectivePolicy: confidenceThreshold must be in [0, 1], got ${threshold}`
225
- );
226
- }
227
- const belowThresholdAction = options.belowThresholdAction ?? "verify";
228
- return {
229
- id: options.id ?? `confidence>=${threshold}`,
230
- decide(point) {
231
- const confidence = point.confidence ?? 0;
232
- return {
233
- action: confidence >= threshold ? "accept" : belowThresholdAction,
234
- confidence,
235
- targetProb: point.targetProb,
236
- qHatChosen: point.qHatChosen,
237
- vHatTarget: point.vHatTarget,
238
- reason: confidence >= threshold ? "confidence threshold passed" : "confidence threshold failed"
239
- };
240
- }
241
- };
179
+ const threshold = options.confidenceThreshold;
180
+ if (!Number.isFinite(threshold) || threshold < 0 || threshold > 1) throw new ValidationError(`thresholdSelectivePolicy: confidenceThreshold must be in [0, 1], got ${threshold}`);
181
+ const belowThresholdAction = options.belowThresholdAction ?? "verify";
182
+ return {
183
+ id: options.id ?? `confidence>=${threshold}`,
184
+ decide(point) {
185
+ const confidence = point.confidence ?? 0;
186
+ return {
187
+ action: confidence >= threshold ? "accept" : belowThresholdAction,
188
+ confidence,
189
+ targetProb: point.targetProb,
190
+ qHatChosen: point.qHatChosen,
191
+ vHatTarget: point.vHatTarget,
192
+ reason: confidence >= threshold ? "confidence threshold passed" : "confidence threshold failed"
193
+ };
194
+ }
195
+ };
242
196
  }
243
197
  function evaluateBeliefSelectivePolicy(points, policy, options = {}) {
244
- const utility = { ...DEFAULT_UTILITY, ...options.utility ?? {} };
245
- const scored = points.filter((point) => point.outcome);
246
- const minN = options.minN ?? 30;
247
- const minAccepted = options.minAccepted ?? 5;
248
- const minUtilityDelta = options.minUtilityDelta ?? 0;
249
- const deltas = [];
250
- const acceptedRewards = [];
251
- const rejectedRewards = [];
252
- let baselineUtility = 0;
253
- let policyUtility = 0;
254
- let accepted = 0;
255
- let acceptedErrors = 0;
256
- for (const point of scored) {
257
- const baseline = acceptUtility(point, utility);
258
- const decision = policy.decide(point);
259
- const candidate = policyDecisionUtility(point, decision.action, utility);
260
- const reward = rewardOf2(point, utility);
261
- baselineUtility += baseline;
262
- policyUtility += candidate;
263
- deltas.push(candidate - baseline);
264
- if (decision.action === "accept") {
265
- accepted++;
266
- acceptedRewards.push(reward);
267
- if (reward < 0) acceptedErrors++;
268
- } else {
269
- rejectedRewards.push(reward);
270
- }
271
- }
272
- const n = scored.length;
273
- const rejected = Math.max(0, n - accepted);
274
- const ci = confidenceInterval(deltas, 0.95, { seed: options.seed ?? 17 });
275
- const reasons = [];
276
- if (n < minN) reasons.push(`need at least ${minN} scored decisions, got ${n}`);
277
- if (accepted < minAccepted)
278
- reasons.push(`need at least ${minAccepted} accepted decisions, got ${accepted}`);
279
- if (ci.lower <= minUtilityDelta) {
280
- reasons.push(`utility CI lower bound ${ci.lower.toFixed(4)} does not clear ${minUtilityDelta}`);
281
- }
282
- const recommendation = n < minN || accepted < minAccepted ? "need_more_data" : ci.lower > minUtilityDelta ? "ship" : "hold";
283
- return {
284
- policyId: policy.id,
285
- n,
286
- accepted,
287
- rejected,
288
- coverage: n > 0 ? accepted / n : 0,
289
- acceptedErrorRate: accepted > 0 ? acceptedErrors / accepted : 0,
290
- baselineUtility,
291
- policyUtility,
292
- utilityDelta: policyUtility - baselineUtility,
293
- utilityCi95: ci,
294
- rejectedMeanReward: rejectedRewards.length > 0 ? mean(rejectedRewards) : null,
295
- recommendation,
296
- reasons
297
- };
198
+ const utility = {
199
+ ...DEFAULT_UTILITY,
200
+ ...options.utility ?? {}
201
+ };
202
+ const scored = points.filter((point) => point.outcome);
203
+ const minN = options.minN ?? 30;
204
+ const minAccepted = options.minAccepted ?? 5;
205
+ const minUtilityDelta = options.minUtilityDelta ?? 0;
206
+ const deltas = [];
207
+ const acceptedRewards = [];
208
+ const rejectedRewards = [];
209
+ let baselineUtility = 0;
210
+ let policyUtility = 0;
211
+ let accepted = 0;
212
+ let acceptedErrors = 0;
213
+ for (const point of scored) {
214
+ const baseline = acceptUtility(point, utility);
215
+ const decision = policy.decide(point);
216
+ const candidate = policyDecisionUtility(point, decision.action, utility);
217
+ const reward = rewardOf(point, utility);
218
+ baselineUtility += baseline;
219
+ policyUtility += candidate;
220
+ deltas.push(candidate - baseline);
221
+ if (decision.action === "accept") {
222
+ accepted++;
223
+ acceptedRewards.push(reward);
224
+ if (reward < 0) acceptedErrors++;
225
+ } else rejectedRewards.push(reward);
226
+ }
227
+ const n = scored.length;
228
+ const rejected = Math.max(0, n - accepted);
229
+ const ci = confidenceInterval(deltas, .95, { seed: options.seed ?? 17 });
230
+ const reasons = [];
231
+ if (n < minN) reasons.push(`need at least ${minN} scored decisions, got ${n}`);
232
+ if (accepted < minAccepted) reasons.push(`need at least ${minAccepted} accepted decisions, got ${accepted}`);
233
+ if (ci.lower <= minUtilityDelta) reasons.push(`utility CI lower bound ${ci.lower.toFixed(4)} does not clear ${minUtilityDelta}`);
234
+ const recommendation = n < minN || accepted < minAccepted ? "need_more_data" : ci.lower > minUtilityDelta ? "ship" : "hold";
235
+ return {
236
+ policyId: policy.id,
237
+ n,
238
+ accepted,
239
+ rejected,
240
+ coverage: n > 0 ? accepted / n : 0,
241
+ acceptedErrorRate: accepted > 0 ? acceptedErrors / accepted : 0,
242
+ baselineUtility,
243
+ policyUtility,
244
+ utilityDelta: policyUtility - baselineUtility,
245
+ utilityCi95: ci,
246
+ rejectedMeanReward: rejectedRewards.length > 0 ? mean$2(rejectedRewards) : null,
247
+ recommendation,
248
+ reasons
249
+ };
298
250
  }
299
251
  function acceptUtility(point, utility) {
300
- return rewardOf2(point, utility) - utility.costWeight * (point.costUsd ?? point.outcome?.costUsd ?? 0);
252
+ return rewardOf(point, utility) - utility.costWeight * (point.costUsd ?? point.outcome?.costUsd ?? 0);
301
253
  }
302
254
  function policyDecisionUtility(point, action, utility) {
303
- if (action === "accept") return acceptUtility(point, utility);
304
- if (action === "verify") return utility.deferUtility - utility.verifyCost;
305
- if (action === "ask") return utility.deferUtility - utility.askCost;
306
- if (action === "retry") return utility.deferUtility - utility.retryCost;
307
- if (action === "stop") return utility.stopUtility;
308
- return utility.deferUtility;
309
- }
310
- function rewardOf2(point, utility) {
311
- const outcome = point.outcome;
312
- if (!outcome) return utility.failureUtility;
313
- if (typeof outcome.reward === "number") return 2 * outcome.reward - 1;
314
- if (typeof outcome.score === "number") return 2 * outcome.score - 1;
315
- if (outcome.success === true) return utility.successUtility;
316
- if (outcome.success === false) return utility.failureUtility;
317
- return utility.failureUtility;
318
- }
319
- function mean(values) {
320
- return values.reduce((sum, value) => sum + value, 0) / values.length;
321
- }
322
-
323
- // src/belief-state/report.ts
255
+ if (action === "accept") return acceptUtility(point, utility);
256
+ if (action === "verify") return utility.deferUtility - utility.verifyCost;
257
+ if (action === "ask") return utility.deferUtility - utility.askCost;
258
+ if (action === "retry") return utility.deferUtility - utility.retryCost;
259
+ if (action === "stop") return utility.stopUtility;
260
+ return utility.deferUtility;
261
+ }
262
+ function rewardOf(point, utility) {
263
+ const outcome = point.outcome;
264
+ if (!outcome) return utility.failureUtility;
265
+ if (typeof outcome.reward === "number") return 2 * outcome.reward - 1;
266
+ if (typeof outcome.score === "number") return 2 * outcome.score - 1;
267
+ if (outcome.success === true) return utility.successUtility;
268
+ if (outcome.success === false) return utility.failureUtility;
269
+ return utility.failureUtility;
270
+ }
271
+ function mean$2(values) {
272
+ return values.reduce((sum, value) => sum + value, 0) / values.length;
273
+ }
274
+ //#endregion
275
+ //#region src/belief-state/report.ts
324
276
  function analyzeBeliefPolicy(options) {
325
- const selective = evaluateBeliefSelectivePolicy(options.points, options.policy, options.selective);
326
- const calibration = calibrateBeliefDecisions(options.points, options.calibration);
327
- const opeTargetPolicy = options.ope?.targetPolicy;
328
- const ope = opeTargetPolicy ? evaluateBeliefOffPolicy(options.points, opeTargetPolicy, options.ope) : null;
329
- const diagnostics = [];
330
- const selectiveStatus = selective.recommendation;
331
- const calibrationStatus = calibration ? "supported" : "unsupported";
332
- const opeRequested = options.requireOpe === true || options.ope !== void 0;
333
- const opeStatus = ope ? ope.support.supported ? "supported" : "unsupported" : opeRequested ? "unsupported" : "not_requested";
334
- if (!calibration) diagnostics.push("calibration unsupported: not enough confidence/outcome pairs");
335
- if (opeRequested && !opeTargetPolicy) diagnostics.push("OPE unsupported: missing target policy");
336
- else if (ope && !ope.support.supported)
337
- diagnostics.push(...ope.support.reasons.map((reason) => `OPE unsupported: ${reason}`));
338
- const status = overallStatus({
339
- selectiveStatus,
340
- hasCalibration: calibration !== null,
341
- opeStatus,
342
- opeRequested
343
- });
344
- return {
345
- policyId: options.policy.id,
346
- n: options.points.length,
347
- status,
348
- selectiveStatus,
349
- calibrationStatus,
350
- opeStatus,
351
- ...ope ? { opeTargetPolicyId: ope.targetPolicyId } : {},
352
- selective,
353
- ...calibration ? { calibration } : {},
354
- ...ope ? { ope } : {},
355
- diagnostics
356
- };
277
+ const selective = evaluateBeliefSelectivePolicy(options.points, options.policy, options.selective);
278
+ const calibration = calibrateBeliefDecisions(options.points, options.calibration);
279
+ const opeTargetPolicy = options.ope?.targetPolicy;
280
+ const ope = opeTargetPolicy ? evaluateBeliefOffPolicy(options.points, opeTargetPolicy, options.ope) : null;
281
+ const diagnostics = [];
282
+ const selectiveStatus = selective.recommendation;
283
+ const calibrationStatus = calibration ? "supported" : "unsupported";
284
+ const opeRequested = options.requireOpe === true || options.ope !== void 0;
285
+ const opeStatus = ope ? ope.support.supported ? "supported" : "unsupported" : opeRequested ? "unsupported" : "not_requested";
286
+ if (!calibration) diagnostics.push("calibration unsupported: not enough confidence/outcome pairs");
287
+ if (opeRequested && !opeTargetPolicy) diagnostics.push("OPE unsupported: missing target policy");
288
+ else if (ope && !ope.support.supported) diagnostics.push(...ope.support.reasons.map((reason) => `OPE unsupported: ${reason}`));
289
+ const status = overallStatus({
290
+ selectiveStatus,
291
+ hasCalibration: calibration !== null,
292
+ opeStatus,
293
+ opeRequested
294
+ });
295
+ return {
296
+ policyId: options.policy.id,
297
+ n: options.points.length,
298
+ status,
299
+ selectiveStatus,
300
+ calibrationStatus,
301
+ opeStatus,
302
+ ...ope ? { opeTargetPolicyId: ope.targetPolicyId } : {},
303
+ selective,
304
+ ...calibration ? { calibration } : {},
305
+ ...ope ? { ope } : {},
306
+ diagnostics
307
+ };
357
308
  }
358
309
  function overallStatus(options) {
359
- if (options.selectiveStatus === "need_more_data" || !options.hasCalibration) {
360
- return "need_more_data";
361
- }
362
- if (options.selectiveStatus === "hold") return "hold";
363
- if (options.opeRequested && options.opeStatus !== "supported") return "hold";
364
- return "ship";
365
- }
366
-
367
- // src/belief-state/code-agent-corpus.ts
368
- var FAILURE_RECOVERY_ACTIONS = ["retry", "verify", "continue", "stop"];
369
- var TARGET_LABELS = {
370
- "failure-recovery": "Failure recovery after tool or patch failure",
371
- "tool-selection": "Tool/action selection",
372
- "graph-completion": "Graph completion decision"
310
+ if (options.selectiveStatus === "need_more_data" || !options.hasCalibration) return "need_more_data";
311
+ if (options.selectiveStatus === "hold") return "hold";
312
+ if (options.opeRequested && options.opeStatus !== "supported") return "hold";
313
+ return "ship";
314
+ }
315
+ //#endregion
316
+ //#region src/belief-state/code-agent-corpus.ts
317
+ const FAILURE_RECOVERY_ACTIONS = [
318
+ "retry",
319
+ "verify",
320
+ "continue",
321
+ "stop"
322
+ ];
323
+ const TARGET_LABELS = {
324
+ "failure-recovery": "Failure recovery after tool or patch failure",
325
+ "tool-selection": "Tool/action selection",
326
+ "graph-completion": "Graph completion decision"
373
327
  };
374
328
  function extractCodeAgentBeliefDecisionPoints(options) {
375
- const entries = options.entries.filter(isRecord);
376
- const diagnostics = [];
377
- const observed = observedActionsFor(options.source, entries, options);
378
- const decisions = [];
379
- for (const action of observed) {
380
- if (action.kind === "tool" || action.kind === "patch") {
381
- decisions.push(toolSelectionDecision(action, options));
382
- }
383
- if (action.kind === "graph-completion") {
384
- decisions.push(graphCompletionDecision(action, options));
385
- }
386
- }
387
- for (const failed of observed) {
388
- if (failed.kind !== "tool" && failed.kind !== "patch" || failed.success !== false) continue;
389
- const next = observed.find(
390
- (candidate) => candidate.stepIndex > failed.stepIndex && (candidate.kind === "tool" || candidate.kind === "patch" || candidate.kind === "terminal")
391
- );
392
- if (!next) {
393
- diagnostics.push({
394
- runId: options.run.runId,
395
- severity: "warning",
396
- reason: `${failed.id}: failed action has no observable follow-up decision`
397
- });
398
- continue;
399
- }
400
- decisions.push(failureRecoveryDecision(failed, next, options));
401
- }
402
- if (decisions.length === 0) {
403
- diagnostics.push({
404
- runId: options.run.runId,
405
- severity: "info",
406
- reason: `no belief decision points extracted from ${options.source} entries`
407
- });
408
- }
409
- return { decisions, diagnostics };
329
+ const entries = options.entries.filter(isRecord$2);
330
+ const diagnostics = [];
331
+ const observed = observedActionsFor(options.source, entries, options);
332
+ const decisions = [];
333
+ for (const action of observed) {
334
+ if (action.kind === "tool" || action.kind === "patch") decisions.push(toolSelectionDecision(action, options));
335
+ if (action.kind === "graph-completion") decisions.push(graphCompletionDecision(action, options));
336
+ }
337
+ for (const failed of observed) {
338
+ if (failed.kind !== "tool" && failed.kind !== "patch" || failed.success !== false) continue;
339
+ const next = observed.find((candidate) => candidate.stepIndex > failed.stepIndex && (candidate.kind === "tool" || candidate.kind === "patch" || candidate.kind === "terminal"));
340
+ if (!next) {
341
+ diagnostics.push({
342
+ runId: options.run.runId,
343
+ severity: "warning",
344
+ reason: `${failed.id}: failed action has no observable follow-up decision`
345
+ });
346
+ continue;
347
+ }
348
+ decisions.push(failureRecoveryDecision(failed, next, options));
349
+ }
350
+ if (decisions.length === 0) diagnostics.push({
351
+ runId: options.run.runId,
352
+ severity: "info",
353
+ reason: `no belief decision points extracted from ${options.source} entries`
354
+ });
355
+ return {
356
+ decisions,
357
+ diagnostics
358
+ };
410
359
  }
411
360
  function inventoryBeliefDecisionPoints(points) {
412
- const byKind = [...groupBy(points, (point) => point.kind).entries()].map(
413
- ([kind, bucketPoints]) => bucketFor(kind, bucketPoints, { kind })
414
- ).sort(sortBuckets);
415
- const byTarget = [...groupBy(points, targetIdOf).entries()].filter((entry) => {
416
- return entry[0] !== void 0;
417
- }).map(([targetId, bucketPoints]) => bucketFor(targetId, bucketPoints, { targetId })).sort(sortBuckets);
418
- const diagnostics = [];
419
- if (points.length === 0) diagnostics.push("no decision points available");
420
- for (const bucket of byTarget) {
421
- if (bucket.withOutcome < bucket.n) {
422
- diagnostics.push(`${bucket.id}: ${bucket.n - bucket.withOutcome} decision(s) missing outcome`);
423
- }
424
- if (bucket.withBehaviorProb < bucket.n || bucket.withTargetProb < bucket.n) {
425
- diagnostics.push(`${bucket.id}: OPE support incomplete`);
426
- }
427
- }
428
- return { n: points.length, byKind, byTarget, diagnostics };
361
+ const byKind = [...groupBy(points, (point) => point.kind).entries()].map(([kind, bucketPoints]) => bucketFor(kind, bucketPoints, { kind })).sort(sortBuckets);
362
+ const byTarget = [...groupBy(points, targetIdOf).entries()].filter((entry) => {
363
+ return entry[0] !== void 0;
364
+ }).map(([targetId, bucketPoints]) => bucketFor(targetId, bucketPoints, { targetId })).sort(sortBuckets);
365
+ const diagnostics = [];
366
+ if (points.length === 0) diagnostics.push("no decision points available");
367
+ for (const bucket of byTarget) {
368
+ if (bucket.withOutcome < bucket.n) diagnostics.push(`${bucket.id}: ${bucket.n - bucket.withOutcome} decision(s) missing outcome`);
369
+ if (bucket.withBehaviorProb < bucket.n || bucket.withTargetProb < bucket.n) diagnostics.push(`${bucket.id}: OPE support incomplete`);
370
+ }
371
+ return {
372
+ n: points.length,
373
+ byKind,
374
+ byTarget,
375
+ diagnostics
376
+ };
429
377
  }
430
378
  function selectBeliefDecisionTarget(points, options = {}) {
431
- const minN = options.minN ?? 10;
432
- const minOutcomeCoverage = options.minOutcomeCoverage ?? 0.8;
433
- const preferredTargets = options.preferredTargets ?? [
434
- "failure-recovery",
435
- "tool-selection",
436
- "graph-completion"
437
- ];
438
- const inventory = inventoryBeliefDecisionPoints(points);
439
- for (const targetId of preferredTargets) {
440
- const support = inventory.byTarget.find((bucket) => bucket.targetId === targetId);
441
- if (!support) continue;
442
- const reasons = [];
443
- if (support.n < minN) reasons.push(`need at least ${minN} decisions, got ${support.n}`);
444
- const outcomeCoverage = support.n > 0 ? support.withOutcome / support.n : 0;
445
- if (outcomeCoverage < minOutcomeCoverage) {
446
- reasons.push(
447
- `outcome coverage ${outcomeCoverage.toFixed(2)} below ${minOutcomeCoverage.toFixed(2)}`
448
- );
449
- }
450
- if (reasons.length > 0) continue;
451
- const targetPoints = points.filter((point) => targetIdOf(point) === targetId);
452
- return {
453
- id: targetId,
454
- label: TARGET_LABELS[targetId],
455
- points: targetPoints,
456
- support,
457
- reasons
458
- };
459
- }
460
- return null;
379
+ const minN = options.minN ?? 10;
380
+ const minOutcomeCoverage = options.minOutcomeCoverage ?? .8;
381
+ const preferredTargets = options.preferredTargets ?? [
382
+ "failure-recovery",
383
+ "tool-selection",
384
+ "graph-completion"
385
+ ];
386
+ const inventory = inventoryBeliefDecisionPoints(points);
387
+ for (const targetId of preferredTargets) {
388
+ const support = inventory.byTarget.find((bucket) => bucket.targetId === targetId);
389
+ if (!support) continue;
390
+ const reasons = [];
391
+ if (support.n < minN) reasons.push(`need at least ${minN} decisions, got ${support.n}`);
392
+ const outcomeCoverage = support.n > 0 ? support.withOutcome / support.n : 0;
393
+ if (outcomeCoverage < minOutcomeCoverage) reasons.push(`outcome coverage ${outcomeCoverage.toFixed(2)} below ${minOutcomeCoverage.toFixed(2)}`);
394
+ if (reasons.length > 0) continue;
395
+ const targetPoints = points.filter((point) => targetIdOf(point) === targetId);
396
+ return {
397
+ id: targetId,
398
+ label: TARGET_LABELS[targetId],
399
+ points: targetPoints,
400
+ support,
401
+ reasons
402
+ };
403
+ }
404
+ return null;
461
405
  }
462
406
  function analyzeBeliefDecisionCorpus(options) {
463
- const inventory = inventoryBeliefDecisionPoints(options.points);
464
- const diagnostics = [...inventory.diagnostics];
465
- const target = options.targetId !== void 0 ? targetSelectionFor(options.points, options.targetId, options) : selectBeliefDecisionTarget(options.points, options);
466
- if (!target) {
467
- diagnostics.push("no decision target has enough support for policy evaluation");
468
- return { inventory, diagnostics };
469
- }
470
- const policy = options.policy ?? thresholdSelectivePolicy({
471
- id: `${target.id}:confidence>=${options.confidenceThreshold ?? 0.5}`,
472
- confidenceThreshold: options.confidenceThreshold ?? 0.5,
473
- belowThresholdAction: "verify"
474
- });
475
- const minN = options.minN ?? 10;
476
- const evaluation = analyzeBeliefPolicy({
477
- points: target.points,
478
- policy,
479
- selective: {
480
- minN,
481
- minAccepted: options.minAccepted ?? Math.min(5, minN),
482
- minUtilityDelta: 0,
483
- ...options.policyOptions?.selective ?? {}
484
- },
485
- calibration: {
486
- minPairs: Math.min(10, minN),
487
- policy,
488
- region: "all",
489
- ...options.policyOptions?.calibration ?? {}
490
- },
491
- ope: {
492
- targetPolicy: embeddedBeliefOpeTargetPolicy(`${target.id}:embedded-target-prob`),
493
- minEffectiveSampleSize: minN,
494
- ...options.policyOptions?.ope ?? {}
495
- },
496
- requireOpe: options.requireOpe ?? true
497
- });
498
- return { inventory, target, policy, evaluation, diagnostics };
407
+ const inventory = inventoryBeliefDecisionPoints(options.points);
408
+ const diagnostics = [...inventory.diagnostics];
409
+ const target = options.targetId !== void 0 ? targetSelectionFor(options.points, options.targetId, options) : selectBeliefDecisionTarget(options.points, options);
410
+ if (!target) {
411
+ diagnostics.push("no decision target has enough support for policy evaluation");
412
+ return {
413
+ inventory,
414
+ diagnostics
415
+ };
416
+ }
417
+ const policy = options.policy ?? thresholdSelectivePolicy({
418
+ id: `${target.id}:confidence>=${options.confidenceThreshold ?? .5}`,
419
+ confidenceThreshold: options.confidenceThreshold ?? .5,
420
+ belowThresholdAction: "verify"
421
+ });
422
+ const minN = options.minN ?? 10;
423
+ return {
424
+ inventory,
425
+ target,
426
+ policy,
427
+ evaluation: analyzeBeliefPolicy({
428
+ points: target.points,
429
+ policy,
430
+ selective: {
431
+ minN,
432
+ minAccepted: options.minAccepted ?? Math.min(5, minN),
433
+ minUtilityDelta: 0,
434
+ ...options.policyOptions?.selective ?? {}
435
+ },
436
+ calibration: {
437
+ minPairs: Math.min(10, minN),
438
+ policy,
439
+ region: "all",
440
+ ...options.policyOptions?.calibration ?? {}
441
+ },
442
+ ope: {
443
+ targetPolicy: embeddedBeliefOpeTargetPolicy(`${target.id}:embedded-target-prob`),
444
+ minEffectiveSampleSize: minN,
445
+ ...options.policyOptions?.ope ?? {}
446
+ },
447
+ requireOpe: options.requireOpe ?? true
448
+ }),
449
+ diagnostics
450
+ };
499
451
  }
500
452
  function observedActionsFor(source, entries, options) {
501
- const observation = options.observation ?? observeCodeAgentSession({ source, entries, sourcePath: options.sourcePath });
502
- if (observation.source !== source) {
503
- throw new Error("code-agent observation source does not match extraction source");
504
- }
505
- return observation.actions.map((action) => observedActionFromSession(action, options));
453
+ const observation = options.observation ?? observeCodeAgentSession({
454
+ source,
455
+ entries,
456
+ sourcePath: options.sourcePath
457
+ });
458
+ if (observation.source !== source) throw new Error("code-agent observation source does not match extraction source");
459
+ return observation.actions.map((action) => observedActionFromSession(action, options));
506
460
  }
507
461
  function observedActionFromSession(action, options) {
508
- return observedAction({
509
- options,
510
- localId: action.id,
511
- stepIndex: action.stepIndex,
512
- kind: action.kind,
513
- action: action.name,
514
- timestamp: action.timestampMs,
515
- success: action.status === "completed" ? true : action.status === "failed" ? false : void 0,
516
- costUsd: action.costUsd,
517
- metadata: { surface: action.surface, status: action.status, ...action.metadata }
518
- });
462
+ return observedAction({
463
+ options,
464
+ localId: action.id,
465
+ stepIndex: action.stepIndex,
466
+ kind: action.kind,
467
+ action: action.name,
468
+ timestamp: action.timestampMs,
469
+ success: action.status === "completed" ? true : action.status === "failed" ? false : void 0,
470
+ costUsd: action.costUsd,
471
+ metadata: {
472
+ surface: action.surface,
473
+ status: action.status,
474
+ ...action.metadata
475
+ }
476
+ });
519
477
  }
520
478
  function toolSelectionDecision(action, options) {
521
- return {
522
- id: `${options.run.runId}:tool-selection:${action.localId}`,
523
- runId: options.run.runId,
524
- scenarioId: options.run.scenarioId,
525
- stepIndex: action.stepIndex,
526
- kind: "tool-select",
527
- chosenAction: action.action,
528
- candidateActions: [action.action],
529
- confidence: 0.65,
530
- costUsd: action.costUsd,
531
- evidence: action.evidence,
532
- outcome: outcomeFromAction(action, options.run),
533
- metadata: {
534
- target: "tool-selection",
535
- source: options.source,
536
- actionKind: action.kind,
537
- confidenceSource: "fixed-observed-action-prior",
538
- ...action.metadata
539
- }
540
- };
479
+ return {
480
+ id: `${options.run.runId}:tool-selection:${action.localId}`,
481
+ runId: options.run.runId,
482
+ scenarioId: options.run.scenarioId,
483
+ stepIndex: action.stepIndex,
484
+ kind: "tool-select",
485
+ chosenAction: action.action,
486
+ candidateActions: [action.action],
487
+ confidence: .65,
488
+ costUsd: action.costUsd,
489
+ evidence: action.evidence,
490
+ outcome: outcomeFromAction(action, options.run),
491
+ metadata: {
492
+ target: "tool-selection",
493
+ source: options.source,
494
+ actionKind: action.kind,
495
+ confidenceSource: "fixed-observed-action-prior",
496
+ ...action.metadata
497
+ }
498
+ };
541
499
  }
542
500
  function graphCompletionDecision(action, options) {
543
- return {
544
- id: `${options.run.runId}:graph-completion:${action.localId}`,
545
- runId: options.run.runId,
546
- scenarioId: options.run.scenarioId,
547
- stepIndex: action.stepIndex,
548
- kind: "stop",
549
- chosenAction: "complete",
550
- candidateActions: ["complete", "continue", "verify"],
551
- confidence: 0.75,
552
- evidence: action.evidence,
553
- outcome: outcomeFromAction(action, options.run),
554
- metadata: {
555
- target: "graph-completion",
556
- source: options.source,
557
- confidenceSource: "fixed-graph-completion-prior",
558
- ...action.metadata
559
- }
560
- };
501
+ return {
502
+ id: `${options.run.runId}:graph-completion:${action.localId}`,
503
+ runId: options.run.runId,
504
+ scenarioId: options.run.scenarioId,
505
+ stepIndex: action.stepIndex,
506
+ kind: "stop",
507
+ chosenAction: "complete",
508
+ candidateActions: [
509
+ "complete",
510
+ "continue",
511
+ "verify"
512
+ ],
513
+ confidence: .75,
514
+ evidence: action.evidence,
515
+ outcome: outcomeFromAction(action, options.run),
516
+ metadata: {
517
+ target: "graph-completion",
518
+ source: options.source,
519
+ confidenceSource: "fixed-graph-completion-prior",
520
+ ...action.metadata
521
+ }
522
+ };
561
523
  }
562
524
  function failureRecoveryDecision(failed, next, options) {
563
- const chosenAction = classifyFailureRecovery(failed, next);
564
- return {
565
- id: `${options.run.runId}:failure-recovery:${failed.localId}`,
566
- runId: options.run.runId,
567
- scenarioId: options.run.scenarioId,
568
- stepIndex: failed.stepIndex,
569
- kind: "retry",
570
- chosenAction,
571
- candidateActions: [...FAILURE_RECOVERY_ACTIONS],
572
- confidence: recoveryConfidence(chosenAction),
573
- evidence: [...failed.evidence, ...next.evidence],
574
- outcome: outcomeFromAction(next, options.run),
575
- metadata: {
576
- target: "failure-recovery",
577
- source: options.source,
578
- failedActionKind: failed.kind,
579
- failedAction: failed.action,
580
- nextActionKind: next.kind,
581
- nextAction: next.action,
582
- confidenceSource: "heuristic-observed-follow-up"
583
- }
584
- };
525
+ const chosenAction = classifyFailureRecovery(failed, next);
526
+ return {
527
+ id: `${options.run.runId}:failure-recovery:${failed.localId}`,
528
+ runId: options.run.runId,
529
+ scenarioId: options.run.scenarioId,
530
+ stepIndex: failed.stepIndex,
531
+ kind: "retry",
532
+ chosenAction,
533
+ candidateActions: [...FAILURE_RECOVERY_ACTIONS],
534
+ confidence: recoveryConfidence(chosenAction),
535
+ evidence: [...failed.evidence, ...next.evidence],
536
+ outcome: outcomeFromAction(next, options.run),
537
+ metadata: {
538
+ target: "failure-recovery",
539
+ source: options.source,
540
+ failedActionKind: failed.kind,
541
+ failedAction: failed.action,
542
+ nextActionKind: next.kind,
543
+ nextAction: next.action,
544
+ confidenceSource: "heuristic-observed-follow-up"
545
+ }
546
+ };
585
547
  }
586
548
  function classifyFailureRecovery(failed, next) {
587
- if (next.kind === "terminal") return "stop";
588
- if (isVerificationAction(next.action)) return "verify";
589
- if (next.kind === failed.kind && next.action === failed.action) return "retry";
590
- return "continue";
549
+ if (next.kind === "terminal") return "stop";
550
+ if (isVerificationAction(next.action)) return "verify";
551
+ if (next.kind === failed.kind && next.action === failed.action) return "retry";
552
+ return "continue";
591
553
  }
592
554
  function recoveryConfidence(action) {
593
- if (action === "verify") return 0.8;
594
- if (action === "retry") return 0.6;
595
- if (action === "stop") return 0.55;
596
- return 0.35;
555
+ if (action === "verify") return .8;
556
+ if (action === "retry") return .6;
557
+ if (action === "stop") return .55;
558
+ return .35;
597
559
  }
598
560
  function isVerificationAction(action) {
599
- const normalized = action.toLowerCase();
600
- return normalized.includes("verify") || normalized.includes("test") || normalized.includes("check") || normalized.includes("lint") || normalized.includes("build") || normalized.includes("typecheck") || normalized.includes("pytest") || normalized.includes("vitest") || normalized.includes("tsc");
561
+ const normalized = action.toLowerCase();
562
+ return normalized.includes("verify") || normalized.includes("test") || normalized.includes("check") || normalized.includes("lint") || normalized.includes("build") || normalized.includes("typecheck") || normalized.includes("pytest") || normalized.includes("vitest") || normalized.includes("tsc");
601
563
  }
602
564
  function outcomeFromAction(action, run) {
603
- const runScore = scoreFromRun(run);
604
- const success = action.success ?? (runScore !== null ? runScore >= 0.5 : void 0);
605
- const score = action.success === void 0 ? runScore ?? void 0 : action.success ? 1 : 0;
606
- if (success === void 0 && score === void 0) return void 0;
607
- return {
608
- ...success !== void 0 ? { success } : {},
609
- ...score !== void 0 ? { score, reward: score } : {},
610
- ...action.costUsd !== void 0 ? { costUsd: action.costUsd } : {},
611
- metadata: {
612
- outcomeSource: action.success === void 0 ? "run-score" : "observed-action-status"
613
- }
614
- };
565
+ const runScore = scoreFromRun(run);
566
+ const success = action.success ?? (runScore !== null ? runScore >= .5 : void 0);
567
+ const score = action.success === void 0 ? runScore ?? void 0 : action.success ? 1 : 0;
568
+ if (success === void 0 && score === void 0) return void 0;
569
+ return {
570
+ ...success !== void 0 ? { success } : {},
571
+ ...score !== void 0 ? {
572
+ score,
573
+ reward: score
574
+ } : {},
575
+ ...action.costUsd !== void 0 ? { costUsd: action.costUsd } : {},
576
+ metadata: { outcomeSource: action.success === void 0 ? "run-score" : "observed-action-status" }
577
+ };
615
578
  }
616
579
  function observedAction(input) {
617
- const id = `${input.options.run.runId}:${input.options.source}:${input.localId}`;
618
- return {
619
- id,
620
- localId: input.localId,
621
- stepIndex: input.stepIndex,
622
- kind: input.kind,
623
- action: input.action,
624
- timestamp: input.timestamp,
625
- success: input.success,
626
- costUsd: input.costUsd,
627
- evidence: [
628
- {
629
- source: "event",
630
- id,
631
- runId: input.options.run.runId,
632
- detail: input.action,
633
- metadata: {
634
- source: input.options.source,
635
- sourcePath: input.options.sourcePath,
636
- ...input.metadata
637
- }
638
- }
639
- ],
640
- metadata: input.metadata ?? {}
641
- };
580
+ const id = `${input.options.run.runId}:${input.options.source}:${input.localId}`;
581
+ return {
582
+ id,
583
+ localId: input.localId,
584
+ stepIndex: input.stepIndex,
585
+ kind: input.kind,
586
+ action: input.action,
587
+ timestamp: input.timestamp,
588
+ success: input.success,
589
+ costUsd: input.costUsd,
590
+ evidence: [{
591
+ source: "event",
592
+ id,
593
+ runId: input.options.run.runId,
594
+ detail: input.action,
595
+ metadata: {
596
+ source: input.options.source,
597
+ sourcePath: input.options.sourcePath,
598
+ ...input.metadata
599
+ }
600
+ }],
601
+ metadata: input.metadata ?? {}
602
+ };
642
603
  }
643
604
  function targetSelectionFor(points, targetId, options) {
644
- const targetPoints = points.filter((point) => targetIdOf(point) === targetId);
645
- if (targetPoints.length === 0) return null;
646
- const support = bucketFor(targetId, targetPoints, { targetId });
647
- const minN = options.minN ?? 10;
648
- const minOutcomeCoverage = options.minOutcomeCoverage ?? 0.8;
649
- const reasons = [];
650
- if (support.n < minN) reasons.push(`need at least ${minN} decisions, got ${support.n}`);
651
- const outcomeCoverage = support.n > 0 ? support.withOutcome / support.n : 0;
652
- if (outcomeCoverage < minOutcomeCoverage) {
653
- reasons.push(
654
- `outcome coverage ${outcomeCoverage.toFixed(2)} below ${minOutcomeCoverage.toFixed(2)}`
655
- );
656
- }
657
- if (reasons.length > 0) return null;
658
- return { id: targetId, label: TARGET_LABELS[targetId], points: targetPoints, support, reasons };
605
+ const targetPoints = points.filter((point) => targetIdOf(point) === targetId);
606
+ if (targetPoints.length === 0) return null;
607
+ const support = bucketFor(targetId, targetPoints, { targetId });
608
+ const minN = options.minN ?? 10;
609
+ const minOutcomeCoverage = options.minOutcomeCoverage ?? .8;
610
+ const reasons = [];
611
+ if (support.n < minN) reasons.push(`need at least ${minN} decisions, got ${support.n}`);
612
+ const outcomeCoverage = support.n > 0 ? support.withOutcome / support.n : 0;
613
+ if (outcomeCoverage < minOutcomeCoverage) reasons.push(`outcome coverage ${outcomeCoverage.toFixed(2)} below ${minOutcomeCoverage.toFixed(2)}`);
614
+ if (reasons.length > 0) return null;
615
+ return {
616
+ id: targetId,
617
+ label: TARGET_LABELS[targetId],
618
+ points: targetPoints,
619
+ support,
620
+ reasons
621
+ };
659
622
  }
660
623
  function bucketFor(id, points, identity) {
661
- const outcomes = points.filter((point) => point.outcome);
662
- const scores = outcomes.map((point) => outcomeScore2(point.outcome)).filter((score) => score !== null);
663
- const confidences = points.map((point) => point.confidence).filter((confidence) => typeof confidence === "number");
664
- const successes = outcomes.filter((point) => point.outcome?.success === true).length;
665
- const successDenominator = outcomes.filter(
666
- (point) => typeof point.outcome?.success === "boolean"
667
- ).length;
668
- return {
669
- id,
670
- ...identity,
671
- n: points.length,
672
- withOutcome: outcomes.length,
673
- withConfidence: confidences.length,
674
- withCandidateActions: points.filter((point) => (point.candidateActions?.length ?? 0) > 0).length,
675
- withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
676
- withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
677
- successRate: successDenominator > 0 ? successes / successDenominator : null,
678
- meanScore: scores.length > 0 ? mean2(scores) : null,
679
- meanConfidence: confidences.length > 0 ? mean2(confidences) : null
680
- };
624
+ const outcomes = points.filter((point) => point.outcome);
625
+ const scores = outcomes.map((point) => outcomeScore(point.outcome)).filter((score) => score !== null);
626
+ const confidences = points.map((point) => point.confidence).filter((confidence) => typeof confidence === "number");
627
+ const successes = outcomes.filter((point) => point.outcome?.success === true).length;
628
+ const successDenominator = outcomes.filter((point) => typeof point.outcome?.success === "boolean").length;
629
+ return {
630
+ id,
631
+ ...identity,
632
+ n: points.length,
633
+ withOutcome: outcomes.length,
634
+ withConfidence: confidences.length,
635
+ withCandidateActions: points.filter((point) => (point.candidateActions?.length ?? 0) > 0).length,
636
+ withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
637
+ withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
638
+ successRate: successDenominator > 0 ? successes / successDenominator : null,
639
+ meanScore: scores.length > 0 ? mean$1(scores) : null,
640
+ meanConfidence: confidences.length > 0 ? mean$1(confidences) : null
641
+ };
681
642
  }
682
643
  function targetIdOf(point) {
683
- const target = point.metadata?.target;
684
- if (target === "failure-recovery" || target === "tool-selection" || target === "graph-completion")
685
- return target;
686
- return void 0;
687
- }
688
- function outcomeScore2(outcome) {
689
- if (!outcome) return null;
690
- if (typeof outcome.score === "number") return outcome.score;
691
- if (typeof outcome.reward === "number") return outcome.reward;
692
- if (outcome.success === true) return 1;
693
- if (outcome.success === false) return 0;
694
- return null;
695
- }
644
+ const target = point.metadata?.target;
645
+ if (target === "failure-recovery" || target === "tool-selection" || target === "graph-completion") return target;
646
+ }
647
+ function outcomeScore(outcome) {
648
+ if (!outcome) return null;
649
+ if (typeof outcome.score === "number") return outcome.score;
650
+ if (typeof outcome.reward === "number") return outcome.reward;
651
+ if (outcome.success === true) return 1;
652
+ if (outcome.success === false) return 0;
653
+ return null;
654
+ }
655
+ /**
656
+ * GATED (`trainingScore`). The number this returns becomes a belief-decision
657
+ * point's `outcome.score` AND its `outcome.reward` — corpus labels, i.e.
658
+ * training data by another name. A run flagged as gamed would otherwise label
659
+ * every decision on its trajectory a success and teach a belief model to
660
+ * predict that the gaming path works.
661
+ */
696
662
  function scoreFromRun(run) {
697
- return trainingScore(run) ?? null;
663
+ return trainingScore(run) ?? null;
698
664
  }
699
665
  function sortBuckets(a, b) {
700
- return b.n - a.n || a.id.localeCompare(b.id);
666
+ return b.n - a.n || a.id.localeCompare(b.id);
701
667
  }
702
668
  function groupBy(values, keyOf) {
703
- const map = /* @__PURE__ */ new Map();
704
- for (const value of values) {
705
- const key = keyOf(value);
706
- const bucket = map.get(key);
707
- if (bucket) bucket.push(value);
708
- else map.set(key, [value]);
709
- }
710
- return map;
711
- }
712
- function mean2(values) {
713
- return values.reduce((sum, value) => sum + value, 0) / values.length;
714
- }
715
- function isRecord(value) {
716
- return value !== null && typeof value === "object" && !Array.isArray(value);
717
- }
718
-
719
- // src/belief-state/research-evidence.ts
669
+ const map = /* @__PURE__ */ new Map();
670
+ for (const value of values) {
671
+ const key = keyOf(value);
672
+ const bucket = map.get(key);
673
+ if (bucket) bucket.push(value);
674
+ else map.set(key, [value]);
675
+ }
676
+ return map;
677
+ }
678
+ function mean$1(values) {
679
+ return values.reduce((sum, value) => sum + value, 0) / values.length;
680
+ }
681
+ function isRecord$2(value) {
682
+ return value !== null && typeof value === "object" && !Array.isArray(value);
683
+ }
684
+ //#endregion
685
+ //#region src/belief-state/research-evidence.ts
720
686
  function buildBeliefDecisionResearchEvidencePacket(options) {
721
- const claimScope = options.claimScope ?? "counterfactual";
722
- const requireOpe = claimScope === "counterfactual";
723
- const analysis = analyzeBeliefDecisionCorpus({
724
- ...options,
725
- requireOpe: options.requireOpe ?? requireOpe
726
- });
727
- const gates = [
728
- corpusGate(analysis),
729
- selectiveGate(analysis),
730
- calibrationGate(analysis),
731
- ...requireOpe ? [opeGate(analysis)] : []
732
- ];
733
- const caveats = unique([
734
- ...gates.flatMap((gate) => gate.caveats),
735
- ...claimScope === "selective" ? ["counterfactual claims excluded: OPE support was not required"] : []
736
- ]);
737
- return {
738
- claimScope,
739
- status: gates.every((gate) => gate.status === "supported") ? "supported" : "blocked",
740
- analysis,
741
- gates,
742
- blockers: unique(gates.flatMap((gate) => gate.blockers)),
743
- caveats
744
- };
687
+ const claimScope = options.claimScope ?? "counterfactual";
688
+ const requireOpe = claimScope === "counterfactual";
689
+ const analysis = analyzeBeliefDecisionCorpus({
690
+ ...options,
691
+ requireOpe: options.requireOpe ?? requireOpe
692
+ });
693
+ const gates = [
694
+ corpusGate(analysis),
695
+ selectiveGate(analysis),
696
+ calibrationGate(analysis),
697
+ ...requireOpe ? [opeGate(analysis)] : []
698
+ ];
699
+ const caveats = unique([...gates.flatMap((gate) => gate.caveats), ...claimScope === "selective" ? ["counterfactual claims excluded: OPE support was not required"] : []]);
700
+ return {
701
+ claimScope,
702
+ status: gates.every((gate) => gate.status === "supported") ? "supported" : "blocked",
703
+ analysis,
704
+ gates,
705
+ blockers: unique(gates.flatMap((gate) => gate.blockers)),
706
+ caveats
707
+ };
745
708
  }
746
709
  function corpusGate(analysis) {
747
- const support = analysis.target?.support;
748
- if (!support) {
749
- return blocked("corpus", "no decision target has enough outcome support");
750
- }
751
- const caveats = support.withBehaviorProb < support.n || support.withTargetProb < support.n ? ["propensity support incomplete; counterfactual claims will require OPE support"] : [];
752
- return { id: "corpus", status: "supported", blockers: [], caveats };
710
+ const support = analysis.target?.support;
711
+ if (!support) return blocked("corpus", "no decision target has enough outcome support");
712
+ return {
713
+ id: "corpus",
714
+ status: "supported",
715
+ blockers: [],
716
+ caveats: support.withBehaviorProb < support.n || support.withTargetProb < support.n ? ["propensity support incomplete; counterfactual claims will require OPE support"] : []
717
+ };
753
718
  }
754
719
  function selectiveGate(analysis) {
755
- const evaluation = analysis.evaluation;
756
- if (!evaluation) return blocked("selective", "no policy evaluation was produced");
757
- if (evaluation.selectiveStatus !== "ship") {
758
- return blocked(
759
- "selective",
760
- ...orDefault(
761
- evaluation.selective.reasons,
762
- `selective status is ${evaluation.selectiveStatus}`
763
- )
764
- );
765
- }
766
- return { id: "selective", status: "supported", blockers: [], caveats: [] };
720
+ const evaluation = analysis.evaluation;
721
+ if (!evaluation) return blocked("selective", "no policy evaluation was produced");
722
+ if (evaluation.selectiveStatus !== "ship") return blocked("selective", ...orDefault(evaluation.selective.reasons, `selective status is ${evaluation.selectiveStatus}`));
723
+ return {
724
+ id: "selective",
725
+ status: "supported",
726
+ blockers: [],
727
+ caveats: []
728
+ };
767
729
  }
768
730
  function calibrationGate(analysis) {
769
- const evaluation = analysis.evaluation;
770
- if (!evaluation) return blocked("calibration", "no policy evaluation was produced");
771
- if (evaluation.calibrationStatus !== "supported") {
772
- return blocked("calibration", "not enough confidence/outcome pairs for calibration");
773
- }
774
- return { id: "calibration", status: "supported", blockers: [], caveats: [] };
731
+ const evaluation = analysis.evaluation;
732
+ if (!evaluation) return blocked("calibration", "no policy evaluation was produced");
733
+ if (evaluation.calibrationStatus !== "supported") return blocked("calibration", "not enough confidence/outcome pairs for calibration");
734
+ return {
735
+ id: "calibration",
736
+ status: "supported",
737
+ blockers: [],
738
+ caveats: []
739
+ };
775
740
  }
776
741
  function opeGate(analysis) {
777
- const evaluation = analysis.evaluation;
778
- if (!evaluation) return blocked("ope", "no policy evaluation was produced");
779
- if (evaluation.opeStatus !== "supported") {
780
- const reasons = evaluation.ope?.support.reasons ?? evaluation.diagnostics.filter((diagnostic) => diagnostic.includes("OPE"));
781
- return blocked("ope", ...orDefault(reasons, "missing OPE support"));
782
- }
783
- return { id: "ope", status: "supported", blockers: [], caveats: [] };
742
+ const evaluation = analysis.evaluation;
743
+ if (!evaluation) return blocked("ope", "no policy evaluation was produced");
744
+ if (evaluation.opeStatus !== "supported") return blocked("ope", ...orDefault(evaluation.ope?.support.reasons ?? evaluation.diagnostics.filter((diagnostic) => diagnostic.includes("OPE")), "missing OPE support"));
745
+ return {
746
+ id: "ope",
747
+ status: "supported",
748
+ blockers: [],
749
+ caveats: []
750
+ };
784
751
  }
785
752
  function blocked(id, ...blockers) {
786
- return { id, status: "blocked", blockers, caveats: [] };
753
+ return {
754
+ id,
755
+ status: "blocked",
756
+ blockers,
757
+ caveats: []
758
+ };
787
759
  }
788
760
  function orDefault(values, fallback) {
789
- return values.length > 0 ? values : [fallback];
761
+ return values.length > 0 ? values : [fallback];
790
762
  }
791
763
  function unique(values) {
792
- return [...new Set(values)];
764
+ return [...new Set(values)];
793
765
  }
794
-
795
- // src/belief-state/code-agent-evidence.ts
766
+ //#endregion
767
+ //#region src/belief-state/code-agent-evidence.ts
796
768
  function buildCodeAgentBeliefEvidenceCorpus(options) {
797
- const { sessions, ...evidenceOptions } = options;
798
- const runs = [];
799
- const metrics = [];
800
- const intakeDiagnostics = [];
801
- const extractionDiagnostics = [];
802
- const decisions = [];
803
- for (const session of sessions) {
804
- const intake = fromCodeAgentBeliefSession(session);
805
- runs.push(...intake.runs);
806
- metrics.push(...intake.metrics);
807
- intakeDiagnostics.push(...intake.diagnostics);
808
- for (const [index, run] of intake.runs.entries()) {
809
- const extraction = extractCodeAgentBeliefDecisionPoints({
810
- source: session.source,
811
- entries: session.entries,
812
- observation: intake.observations[index],
813
- run,
814
- sourcePath: session.sourcePath
815
- });
816
- decisions.push(...extraction.decisions);
817
- extractionDiagnostics.push(...extraction.diagnostics);
818
- }
819
- }
820
- const evidence = buildBeliefDecisionResearchEvidencePacket({
821
- ...evidenceOptions,
822
- points: decisions
823
- });
824
- return {
825
- runs,
826
- metrics,
827
- intakeDiagnostics,
828
- extractionDiagnostics,
829
- decisions,
830
- inventory: inventoryBeliefDecisionPoints(decisions),
831
- evidence
832
- };
769
+ const { sessions, ...evidenceOptions } = options;
770
+ const runs = [];
771
+ const metrics = [];
772
+ const intakeDiagnostics = [];
773
+ const extractionDiagnostics = [];
774
+ const decisions = [];
775
+ for (const session of sessions) {
776
+ const intake = fromCodeAgentBeliefSession(session);
777
+ runs.push(...intake.runs);
778
+ metrics.push(...intake.metrics);
779
+ intakeDiagnostics.push(...intake.diagnostics);
780
+ for (const [index, run] of intake.runs.entries()) {
781
+ const extraction = extractCodeAgentBeliefDecisionPoints({
782
+ source: session.source,
783
+ entries: session.entries,
784
+ observation: intake.observations[index],
785
+ run,
786
+ sourcePath: session.sourcePath
787
+ });
788
+ decisions.push(...extraction.decisions);
789
+ extractionDiagnostics.push(...extraction.diagnostics);
790
+ }
791
+ }
792
+ const evidence = buildBeliefDecisionResearchEvidencePacket({
793
+ ...evidenceOptions,
794
+ points: decisions
795
+ });
796
+ return {
797
+ runs,
798
+ metrics,
799
+ intakeDiagnostics,
800
+ extractionDiagnostics,
801
+ decisions,
802
+ inventory: inventoryBeliefDecisionPoints(decisions),
803
+ evidence
804
+ };
833
805
  }
834
806
  function fromCodeAgentBeliefSession(session) {
835
- switch (session.source) {
836
- case "codex":
837
- return fromCodexSession(session);
838
- case "claude-code":
839
- return fromClaudeCodeSession(session);
840
- case "opencode":
841
- return fromOpenCodeSession(session);
842
- case "kimi-code":
843
- return fromKimiCodeSession(session);
844
- case "pi":
845
- return fromPiSession(session);
846
- }
847
- }
848
-
849
- // src/belief-state/types.ts
850
- var BELIEF_DECISION_KINDS = [
851
- "continue",
852
- "verify",
853
- "ask",
854
- "retry",
855
- "stop",
856
- "memory-write",
857
- "memory-read",
858
- "tool-select",
859
- "skill-select",
860
- "workflow-select",
861
- "surface-promote"
807
+ switch (session.source) {
808
+ case "codex": return fromCodexSession(session);
809
+ case "claude-code": return fromClaudeCodeSession(session);
810
+ case "opencode": return fromOpenCodeSession(session);
811
+ case "kimi-code": return fromKimiCodeSession(session);
812
+ case "pi": return fromPiSession(session);
813
+ }
814
+ }
815
+ //#endregion
816
+ //#region src/belief-state/types.ts
817
+ const BELIEF_DECISION_KINDS = [
818
+ "continue",
819
+ "verify",
820
+ "ask",
821
+ "retry",
822
+ "stop",
823
+ "memory-write",
824
+ "memory-read",
825
+ "tool-select",
826
+ "skill-select",
827
+ "workflow-select",
828
+ "surface-promote"
862
829
  ];
863
- var BELIEF_EVIDENCE_SOURCES = [
864
- "run",
865
- "span",
866
- "event",
867
- "finding",
868
- "memory",
869
- "knowledge",
870
- "policy"
830
+ const BELIEF_EVIDENCE_SOURCES = [
831
+ "run",
832
+ "span",
833
+ "event",
834
+ "finding",
835
+ "memory",
836
+ "knowledge",
837
+ "policy"
871
838
  ];
872
- var BELIEF_EVIDENCE_QUALITIES = [
873
- "direct",
874
- "derived",
875
- "self-reported",
876
- "unverified",
877
- "stale",
878
- "contradicted"
839
+ const BELIEF_EVIDENCE_QUALITIES = [
840
+ "direct",
841
+ "derived",
842
+ "self-reported",
843
+ "unverified",
844
+ "stale",
845
+ "contradicted"
879
846
  ];
880
- var BELIEF_EVALUATION_CRITERIA = [
881
- {
882
- id: "capture-integrity",
883
- label: "Capture integrity",
884
- reasonCodes: ["trace-missing", "run-record-missing", "backend-integrity-missing"]
885
- },
886
- {
887
- id: "decision-completeness",
888
- label: "Decision completeness",
889
- reasonCodes: [
890
- "candidate-actions-missing",
891
- "chosen-action-missing",
892
- "decision-evidence-missing"
893
- ]
894
- },
895
- {
896
- id: "evidence-quality",
897
- label: "Evidence quality",
898
- reasonCodes: [
899
- "evidence-stale",
900
- "evidence-contradictory",
901
- "evidence-unverified",
902
- "evidence-self-reported"
903
- ]
904
- },
905
- {
906
- id: "outcome-quality",
907
- label: "Outcome quality",
908
- reasonCodes: ["outcome-missing", "outcome-delayed", "cost-missing"]
909
- },
910
- {
911
- id: "calibration",
912
- label: "Calibration",
913
- reasonCodes: ["confidence-missing", "calibration-unsupported", "calibration-gap-high"]
914
- },
915
- {
916
- id: "accepted-region-risk",
917
- label: "Accepted-region risk",
918
- reasonCodes: ["accepted-error-high", "coverage-too-low"]
919
- },
920
- {
921
- id: "policy-value",
922
- label: "Policy value",
923
- reasonCodes: ["utility-lift-missing", "baseline-dominates", "cost-too-high"]
924
- },
925
- {
926
- id: "ope-support",
927
- label: "OPE support",
928
- reasonCodes: [
929
- "behavior-propensity-missing",
930
- "behavior-propensity-invalid",
931
- "target-propensity-missing",
932
- "target-propensity-invalid",
933
- "effective-sample-size-low",
934
- "importance-weight-high"
935
- ]
936
- },
937
- {
938
- id: "memory-health",
939
- label: "Memory health",
940
- reasonCodes: [
941
- "memory-stale",
942
- "memory-poisoning-risk",
943
- "context-bloat",
944
- "memory-write-unverified"
945
- ]
946
- },
947
- {
948
- id: "surface-attribution",
949
- label: "Surface attribution",
950
- reasonCodes: ["surface-claim-unsupported", "causal-attribution-missing"]
951
- },
952
- {
953
- id: "generalization",
954
- label: "Generalization",
955
- reasonCodes: [
956
- "split-missing",
957
- "holdout-regression",
958
- "task-family-coverage-low",
959
- "leakage-risk"
960
- ]
961
- },
962
- {
963
- id: "promotion",
964
- label: "Promotion",
965
- reasonCodes: ["negative-control-failed", "promotion-gate-failed", "human-review-required"]
966
- }
847
+ const BELIEF_EVALUATION_CRITERIA = [
848
+ {
849
+ id: "capture-integrity",
850
+ label: "Capture integrity",
851
+ reasonCodes: [
852
+ "trace-missing",
853
+ "run-record-missing",
854
+ "backend-integrity-missing"
855
+ ]
856
+ },
857
+ {
858
+ id: "decision-completeness",
859
+ label: "Decision completeness",
860
+ reasonCodes: [
861
+ "candidate-actions-missing",
862
+ "chosen-action-missing",
863
+ "decision-evidence-missing"
864
+ ]
865
+ },
866
+ {
867
+ id: "evidence-quality",
868
+ label: "Evidence quality",
869
+ reasonCodes: [
870
+ "evidence-stale",
871
+ "evidence-contradictory",
872
+ "evidence-unverified",
873
+ "evidence-self-reported"
874
+ ]
875
+ },
876
+ {
877
+ id: "outcome-quality",
878
+ label: "Outcome quality",
879
+ reasonCodes: [
880
+ "outcome-missing",
881
+ "outcome-delayed",
882
+ "cost-missing"
883
+ ]
884
+ },
885
+ {
886
+ id: "calibration",
887
+ label: "Calibration",
888
+ reasonCodes: [
889
+ "confidence-missing",
890
+ "calibration-unsupported",
891
+ "calibration-gap-high"
892
+ ]
893
+ },
894
+ {
895
+ id: "accepted-region-risk",
896
+ label: "Accepted-region risk",
897
+ reasonCodes: ["accepted-error-high", "coverage-too-low"]
898
+ },
899
+ {
900
+ id: "policy-value",
901
+ label: "Policy value",
902
+ reasonCodes: [
903
+ "utility-lift-missing",
904
+ "baseline-dominates",
905
+ "cost-too-high"
906
+ ]
907
+ },
908
+ {
909
+ id: "ope-support",
910
+ label: "OPE support",
911
+ reasonCodes: [
912
+ "behavior-propensity-missing",
913
+ "behavior-propensity-invalid",
914
+ "target-propensity-missing",
915
+ "target-propensity-invalid",
916
+ "effective-sample-size-low",
917
+ "importance-weight-high"
918
+ ]
919
+ },
920
+ {
921
+ id: "memory-health",
922
+ label: "Memory health",
923
+ reasonCodes: [
924
+ "memory-stale",
925
+ "memory-poisoning-risk",
926
+ "context-bloat",
927
+ "memory-write-unverified"
928
+ ]
929
+ },
930
+ {
931
+ id: "surface-attribution",
932
+ label: "Surface attribution",
933
+ reasonCodes: ["surface-claim-unsupported", "causal-attribution-missing"]
934
+ },
935
+ {
936
+ id: "generalization",
937
+ label: "Generalization",
938
+ reasonCodes: [
939
+ "split-missing",
940
+ "holdout-regression",
941
+ "task-family-coverage-low",
942
+ "leakage-risk"
943
+ ]
944
+ },
945
+ {
946
+ id: "promotion",
947
+ label: "Promotion",
948
+ reasonCodes: [
949
+ "negative-control-failed",
950
+ "promotion-gate-failed",
951
+ "human-review-required"
952
+ ]
953
+ }
967
954
  ];
968
955
  function isBeliefDecisionKind(value) {
969
- return typeof value === "string" && BELIEF_DECISION_KINDS.includes(value);
956
+ return typeof value === "string" && BELIEF_DECISION_KINDS.includes(value);
970
957
  }
971
958
  function isBeliefEvidenceSource(value) {
972
- return typeof value === "string" && BELIEF_EVIDENCE_SOURCES.includes(value);
973
- }
974
-
975
- // src/belief-state/extract.ts
976
- var DECISION_MARKERS = /* @__PURE__ */ new Set(["belief_decision", "belief.decision", "decision_point"]);
959
+ return typeof value === "string" && BELIEF_EVIDENCE_SOURCES.includes(value);
960
+ }
961
+ //#endregion
962
+ //#region src/belief-state/extract.ts
963
+ const DECISION_MARKERS = /* @__PURE__ */ new Set([
964
+ "belief_decision",
965
+ "belief.decision",
966
+ "decision_point"
967
+ ]);
977
968
  async function extractBeliefDecisionPoints(store, options = {}) {
978
- const runs = options.runIds ? (await Promise.all(options.runIds.map((runId) => store.getRun(runId)))).filter(Boolean) : await store.listRuns();
979
- const decisions = [];
980
- const diagnostics = [];
981
- for (const run of runs) {
982
- if (!run) continue;
983
- const events = await store.events({ runId: run.runId });
984
- const spans = await store.spans({ runId: run.runId });
985
- const spanIds = new Set(spans.map((span) => span.spanId));
986
- let stepIndex = 0;
987
- for (const event of [...events].sort((a, b) => a.timestamp - b.timestamp)) {
988
- const parsed = parseDecisionEvent(event, {
989
- scenarioId: run.scenarioId,
990
- stepIndex,
991
- spanExists: event.spanId ? spanIds.has(event.spanId) : false
992
- });
993
- if (!parsed) continue;
994
- if ("diagnostic" in parsed) {
995
- diagnostics.push(parsed.diagnostic);
996
- continue;
997
- }
998
- decisions.push(parsed.decision);
999
- stepIndex++;
1000
- }
1001
- }
1002
- return { decisions, diagnostics };
969
+ const runs = options.runIds ? (await Promise.all(options.runIds.map((runId) => store.getRun(runId)))).filter(Boolean) : await store.listRuns();
970
+ const decisions = [];
971
+ const diagnostics = [];
972
+ for (const run of runs) {
973
+ if (!run) continue;
974
+ const events = await store.events({ runId: run.runId });
975
+ const spans = await store.spans({ runId: run.runId });
976
+ const spanIds = new Set(spans.map((span) => span.spanId));
977
+ let stepIndex = 0;
978
+ for (const event of [...events].sort((a, b) => a.timestamp - b.timestamp)) {
979
+ const parsed = parseDecisionEvent(event, {
980
+ scenarioId: run.scenarioId,
981
+ stepIndex,
982
+ spanExists: event.spanId ? spanIds.has(event.spanId) : false
983
+ });
984
+ if (!parsed) continue;
985
+ if ("diagnostic" in parsed) {
986
+ diagnostics.push(parsed.diagnostic);
987
+ continue;
988
+ }
989
+ decisions.push(parsed.decision);
990
+ stepIndex++;
991
+ }
992
+ }
993
+ return {
994
+ decisions,
995
+ diagnostics
996
+ };
1003
997
  }
1004
998
  function parseDecisionEvent(event, context) {
1005
- const payload = event.payload;
1006
- const marker = stringField(payload, "kind") ?? stringField(payload, "type");
1007
- if (!marker || !DECISION_MARKERS.has(marker)) return null;
1008
- const decisionKind = stringField(payload, "decisionKind");
1009
- if (!isBeliefDecisionKind(decisionKind)) {
1010
- return {
1011
- diagnostic: {
1012
- runId: event.runId,
1013
- eventId: event.eventId,
1014
- severity: "warning",
1015
- reason: `belief decision event has unsupported decisionKind "${decisionKind ?? ""}"`
1016
- }
1017
- };
1018
- }
1019
- const chosenAction = stringField(payload, "chosenAction");
1020
- if (!chosenAction) {
1021
- return {
1022
- diagnostic: {
1023
- runId: event.runId,
1024
- eventId: event.eventId,
1025
- severity: "warning",
1026
- reason: "belief decision event is missing chosenAction"
1027
- }
1028
- };
1029
- }
1030
- const evidence = [
1031
- {
1032
- source: "event",
1033
- id: event.eventId,
1034
- runId: event.runId,
1035
- eventId: event.eventId,
1036
- quality: "direct"
1037
- }
1038
- ];
1039
- if (event.spanId && context.spanExists) {
1040
- evidence.push({
1041
- source: "span",
1042
- id: event.spanId,
1043
- runId: event.runId,
1044
- spanId: event.spanId,
1045
- quality: "direct"
1046
- });
1047
- }
1048
- return {
1049
- decision: {
1050
- id: stringField(payload, "id") ?? event.eventId,
1051
- runId: event.runId,
1052
- scenarioId: stringField(payload, "scenarioId") ?? context.scenarioId,
1053
- stepIndex: numberField(payload, "stepIndex") ?? context.stepIndex,
1054
- kind: decisionKind,
1055
- chosenAction,
1056
- candidateActions: stringArrayField(payload, "candidateActions"),
1057
- confidence: finiteUnitField(payload, "confidence"),
1058
- behaviorProb: numberField(payload, "behaviorProb"),
1059
- targetProb: numberField(payload, "targetProb"),
1060
- qHatChosen: finiteUnitField(payload, "qHatChosen"),
1061
- vHatTarget: finiteUnitField(payload, "vHatTarget"),
1062
- costUsd: nonNegativeNumberField(payload, "costUsd"),
1063
- evidence,
1064
- outcome: parseOutcome(payload),
1065
- metadata: recordField(payload, "metadata")
1066
- }
1067
- };
999
+ const payload = event.payload;
1000
+ const marker = stringField(payload, "kind") ?? stringField(payload, "type");
1001
+ if (!marker || !DECISION_MARKERS.has(marker)) return null;
1002
+ const decisionKind = stringField(payload, "decisionKind");
1003
+ if (!isBeliefDecisionKind(decisionKind)) return { diagnostic: {
1004
+ runId: event.runId,
1005
+ eventId: event.eventId,
1006
+ severity: "warning",
1007
+ reason: `belief decision event has unsupported decisionKind "${decisionKind ?? ""}"`
1008
+ } };
1009
+ const chosenAction = stringField(payload, "chosenAction");
1010
+ if (!chosenAction) return { diagnostic: {
1011
+ runId: event.runId,
1012
+ eventId: event.eventId,
1013
+ severity: "warning",
1014
+ reason: "belief decision event is missing chosenAction"
1015
+ } };
1016
+ const evidence = [{
1017
+ source: "event",
1018
+ id: event.eventId,
1019
+ runId: event.runId,
1020
+ eventId: event.eventId,
1021
+ quality: "direct"
1022
+ }];
1023
+ if (event.spanId && context.spanExists) evidence.push({
1024
+ source: "span",
1025
+ id: event.spanId,
1026
+ runId: event.runId,
1027
+ spanId: event.spanId,
1028
+ quality: "direct"
1029
+ });
1030
+ return { decision: {
1031
+ id: stringField(payload, "id") ?? event.eventId,
1032
+ runId: event.runId,
1033
+ scenarioId: stringField(payload, "scenarioId") ?? context.scenarioId,
1034
+ stepIndex: numberField(payload, "stepIndex") ?? context.stepIndex,
1035
+ kind: decisionKind,
1036
+ chosenAction,
1037
+ candidateActions: stringArrayField(payload, "candidateActions"),
1038
+ confidence: finiteUnitField(payload, "confidence"),
1039
+ behaviorProb: numberField(payload, "behaviorProb"),
1040
+ targetProb: numberField(payload, "targetProb"),
1041
+ qHatChosen: finiteUnitField(payload, "qHatChosen"),
1042
+ vHatTarget: finiteUnitField(payload, "vHatTarget"),
1043
+ costUsd: nonNegativeNumberField(payload, "costUsd"),
1044
+ evidence,
1045
+ outcome: parseOutcome(payload),
1046
+ metadata: recordField(payload, "metadata")
1047
+ } };
1068
1048
  }
1069
1049
  function parseOutcome(payload) {
1070
- const value = recordField(payload, "outcome");
1071
- if (!value) return void 0;
1072
- return {
1073
- success: typeof value.success === "boolean" ? value.success : void 0,
1074
- score: finiteUnitField(value, "score"),
1075
- reward: finiteUnitField(value, "reward"),
1076
- costUsd: nonNegativeNumberField(value, "costUsd"),
1077
- observedAt: stringField(value, "observedAt"),
1078
- metadata: recordField(value, "metadata")
1079
- };
1050
+ const value = recordField(payload, "outcome");
1051
+ if (!value) return void 0;
1052
+ return {
1053
+ success: typeof value.success === "boolean" ? value.success : void 0,
1054
+ score: finiteUnitField(value, "score"),
1055
+ reward: finiteUnitField(value, "reward"),
1056
+ costUsd: nonNegativeNumberField(value, "costUsd"),
1057
+ observedAt: stringField(value, "observedAt"),
1058
+ metadata: recordField(value, "metadata")
1059
+ };
1080
1060
  }
1081
1061
  function stringField(obj, key) {
1082
- const value = obj[key];
1083
- return typeof value === "string" && value.length > 0 ? value : void 0;
1062
+ const value = obj[key];
1063
+ return typeof value === "string" && value.length > 0 ? value : void 0;
1084
1064
  }
1085
1065
  function numberField(obj, key) {
1086
- const value = obj[key];
1087
- return typeof value === "number" && Number.isFinite(value) ? value : void 0;
1066
+ const value = obj[key];
1067
+ return typeof value === "number" && Number.isFinite(value) ? value : void 0;
1088
1068
  }
1089
1069
  function finiteUnitField(obj, key) {
1090
- const value = numberField(obj, key);
1091
- return value === void 0 ? void 0 : Math.max(0, Math.min(1, value));
1070
+ const value = numberField(obj, key);
1071
+ return value === void 0 ? void 0 : Math.max(0, Math.min(1, value));
1092
1072
  }
1093
1073
  function nonNegativeNumberField(obj, key) {
1094
- const value = numberField(obj, key);
1095
- return value === void 0 ? void 0 : Math.max(0, value);
1074
+ const value = numberField(obj, key);
1075
+ return value === void 0 ? void 0 : Math.max(0, value);
1096
1076
  }
1097
1077
  function stringArrayField(obj, key) {
1098
- const value = obj[key];
1099
- if (!Array.isArray(value)) return void 0;
1100
- const strings = value.filter(
1101
- (item) => typeof item === "string" && item.length > 0
1102
- );
1103
- return strings.length > 0 ? strings : void 0;
1078
+ const value = obj[key];
1079
+ if (!Array.isArray(value)) return void 0;
1080
+ const strings = value.filter((item) => typeof item === "string" && item.length > 0);
1081
+ return strings.length > 0 ? strings : void 0;
1104
1082
  }
1105
1083
  function recordField(obj, key) {
1106
- const value = obj[key];
1107
- if (!value || typeof value !== "object" || Array.isArray(value)) return void 0;
1108
- return value;
1109
- }
1110
-
1111
- // src/belief-state/runtime-hooks.ts
1112
- var DEFAULT_MAX_CONTEXT_CHARS = 12e3;
1113
- var DEFAULT_PAYLOAD_PREVIEW_CHARS = 2e3;
1084
+ const value = obj[key];
1085
+ if (!value || typeof value !== "object" || Array.isArray(value)) return void 0;
1086
+ return value;
1087
+ }
1088
+ //#endregion
1089
+ //#region src/belief-state/runtime-hooks.ts
1090
+ const DEFAULT_MAX_CONTEXT_CHARS$1 = 12e3;
1091
+ const DEFAULT_PAYLOAD_PREVIEW_CHARS = 2e3;
1114
1092
  function runtimeDecisionPointToBeliefShadowProbeInput(point, options) {
1115
- const diagnostics = [];
1116
- const decisionKind = resolveDecisionKind(point, options.decisionKind, diagnostics);
1117
- if (!decisionKind) return { diagnostics };
1118
- const lifecycleEvidence = runtimeHookEventsToEvidenceRefs(point, options);
1119
- const evidence = [...point.evidence ?? [], ...lifecycleEvidence];
1120
- return {
1121
- input: {
1122
- probeId: options.probeId,
1123
- decisionId: point.id,
1124
- runId: point.runId,
1125
- scenarioId: point.scenarioId,
1126
- stepIndex: point.stepIndex,
1127
- decisionKind,
1128
- candidateActions: uniqueStrings(point.candidateActions ?? []),
1129
- evidence: evidence.map((ref) => ({
1130
- id: ref.id,
1131
- source: ref.source,
1132
- ...options.includeEvidenceDetail && ref.detail ? { detail: ref.detail } : {},
1133
- ...ref.quality ? { quality: ref.quality } : {}
1134
- })),
1135
- context: trimText(point.context, options.maxContextChars),
1136
- metadata: mergeMetadata(point.metadata, lifecycleMetadata(lifecycleEvidence))
1137
- },
1138
- diagnostics
1139
- };
1093
+ const diagnostics = [];
1094
+ const decisionKind = resolveDecisionKind(point, options.decisionKind, diagnostics);
1095
+ if (!decisionKind) return { diagnostics };
1096
+ const lifecycleEvidence = runtimeHookEventsToEvidenceRefs(point, options);
1097
+ const evidence = [...point.evidence ?? [], ...lifecycleEvidence];
1098
+ return {
1099
+ input: {
1100
+ probeId: options.probeId,
1101
+ decisionId: point.id,
1102
+ runId: point.runId,
1103
+ scenarioId: point.scenarioId,
1104
+ stepIndex: point.stepIndex,
1105
+ decisionKind,
1106
+ candidateActions: uniqueStrings$1(point.candidateActions ?? []),
1107
+ evidence: evidence.map((ref) => ({
1108
+ id: ref.id,
1109
+ source: ref.source,
1110
+ ...options.includeEvidenceDetail && ref.detail ? { detail: ref.detail } : {},
1111
+ ...ref.quality ? { quality: ref.quality } : {}
1112
+ })),
1113
+ context: trimText$1(point.context, options.maxContextChars),
1114
+ metadata: mergeMetadata(point.metadata, lifecycleMetadata(lifecycleEvidence))
1115
+ },
1116
+ diagnostics
1117
+ };
1140
1118
  }
1141
1119
  function runtimeDecisionPointToBeliefDecisionPoint(point, options) {
1142
- const diagnostics = [];
1143
- const decisionKind = resolveDecisionKind(point, options.decisionKind, diagnostics);
1144
- const chosenAction = stringOrUndefined(options.chosenAction);
1145
- if (!chosenAction) {
1146
- diagnostics.push({
1147
- decisionId: point.id,
1148
- severity: "error",
1149
- reason: "missing chosenAction"
1150
- });
1151
- }
1152
- const candidateActions = uniqueStrings(point.candidateActions ?? []);
1153
- if (chosenAction && candidateActions.length > 0 && !candidateActions.includes(chosenAction)) {
1154
- diagnostics.push({
1155
- decisionId: point.id,
1156
- severity: "warning",
1157
- reason: `chosenAction ${chosenAction} is not in candidateActions`
1158
- });
1159
- }
1160
- if (!decisionKind || !chosenAction) return { diagnostics };
1161
- const lifecycleEvidence = runtimeHookEventsToEvidenceRefs(point, options);
1162
- const evidence = [...point.evidence ?? [], ...lifecycleEvidence];
1163
- return {
1164
- point: {
1165
- id: point.id,
1166
- runId: point.runId,
1167
- scenarioId: point.scenarioId,
1168
- stepIndex: point.stepIndex,
1169
- kind: decisionKind,
1170
- chosenAction,
1171
- candidateActions,
1172
- confidence: unitProbabilityOrUndefined(options.confidence),
1173
- behaviorProb: finiteNumberOrUndefined(options.behaviorProb),
1174
- targetProb: finiteNumberOrUndefined(options.targetProb),
1175
- qHatChosen: options.qHatChosen === null ? null : unitProbabilityOrUndefined(options.qHatChosen),
1176
- vHatTarget: options.vHatTarget === null ? null : unitProbabilityOrUndefined(options.vHatTarget),
1177
- costUsd: nonNegativeNumberOrUndefined(options.costUsd),
1178
- evidence: evidence.map((ref) => runtimeEvidenceToBeliefEvidence(ref, point)),
1179
- outcome: options.outcome,
1180
- metadata: mergeMetadata(
1181
- mergeMetadata(point.metadata, lifecycleMetadata(lifecycleEvidence)),
1182
- options.metadata
1183
- )
1184
- },
1185
- diagnostics
1186
- };
1120
+ const diagnostics = [];
1121
+ const decisionKind = resolveDecisionKind(point, options.decisionKind, diagnostics);
1122
+ const chosenAction = stringOrUndefined(options.chosenAction);
1123
+ if (!chosenAction) diagnostics.push({
1124
+ decisionId: point.id,
1125
+ severity: "error",
1126
+ reason: "missing chosenAction"
1127
+ });
1128
+ const candidateActions = uniqueStrings$1(point.candidateActions ?? []);
1129
+ if (chosenAction && candidateActions.length > 0 && !candidateActions.includes(chosenAction)) diagnostics.push({
1130
+ decisionId: point.id,
1131
+ severity: "warning",
1132
+ reason: `chosenAction ${chosenAction} is not in candidateActions`
1133
+ });
1134
+ if (!decisionKind || !chosenAction) return { diagnostics };
1135
+ const lifecycleEvidence = runtimeHookEventsToEvidenceRefs(point, options);
1136
+ const evidence = [...point.evidence ?? [], ...lifecycleEvidence];
1137
+ return {
1138
+ point: {
1139
+ id: point.id,
1140
+ runId: point.runId,
1141
+ scenarioId: point.scenarioId,
1142
+ stepIndex: point.stepIndex,
1143
+ kind: decisionKind,
1144
+ chosenAction,
1145
+ candidateActions,
1146
+ confidence: unitProbabilityOrUndefined(options.confidence),
1147
+ behaviorProb: finiteNumberOrUndefined(options.behaviorProb),
1148
+ targetProb: finiteNumberOrUndefined(options.targetProb),
1149
+ qHatChosen: options.qHatChosen === null ? null : unitProbabilityOrUndefined(options.qHatChosen),
1150
+ vHatTarget: options.vHatTarget === null ? null : unitProbabilityOrUndefined(options.vHatTarget),
1151
+ costUsd: nonNegativeNumberOrUndefined(options.costUsd),
1152
+ evidence: evidence.map((ref) => runtimeEvidenceToBeliefEvidence(ref, point)),
1153
+ outcome: options.outcome,
1154
+ metadata: mergeMetadata(mergeMetadata(point.metadata, lifecycleMetadata(lifecycleEvidence)), options.metadata)
1155
+ },
1156
+ diagnostics
1157
+ };
1187
1158
  }
1188
1159
  function createBeliefRuntimeHookCollector(defaults) {
1189
- const decisions = [];
1190
- const events = [];
1191
- return {
1192
- hooks: {
1193
- onEvent: (event) => {
1194
- events.push(snapshotRuntimeHookEvent(event));
1195
- },
1196
- onDecisionPoint: (point) => {
1197
- decisions.push(snapshotRuntimeDecisionPoint(point));
1198
- }
1199
- },
1200
- decisions,
1201
- events,
1202
- toShadowProbeInputs: (options = {}) => {
1203
- const inputs = [];
1204
- const diagnostics = [];
1205
- const includeLifecycleEvidence = options.includeLifecycleEvidence ?? defaults.includeLifecycleEvidence;
1206
- for (const point of decisions) {
1207
- const report = runtimeDecisionPointToBeliefShadowProbeInput(point, {
1208
- ...defaults,
1209
- ...options,
1210
- includeLifecycleEvidence,
1211
- lifecycleEvents: includeLifecycleEvidence === false ? void 0 : options.lifecycleEvents ?? defaults.lifecycleEvents ?? events
1212
- });
1213
- if (report.input) inputs.push(report.input);
1214
- diagnostics.push(...report.diagnostics);
1215
- }
1216
- return { inputs, diagnostics };
1217
- },
1218
- clear: () => {
1219
- decisions.length = 0;
1220
- events.length = 0;
1221
- }
1222
- };
1160
+ const decisions = [];
1161
+ const events = [];
1162
+ return {
1163
+ hooks: {
1164
+ onEvent: (event) => {
1165
+ events.push(snapshotRuntimeHookEvent(event));
1166
+ },
1167
+ onDecisionPoint: (point) => {
1168
+ decisions.push(snapshotRuntimeDecisionPoint(point));
1169
+ }
1170
+ },
1171
+ decisions,
1172
+ events,
1173
+ toShadowProbeInputs: (options = {}) => {
1174
+ const inputs = [];
1175
+ const diagnostics = [];
1176
+ const includeLifecycleEvidence = options.includeLifecycleEvidence ?? defaults.includeLifecycleEvidence;
1177
+ for (const point of decisions) {
1178
+ const report = runtimeDecisionPointToBeliefShadowProbeInput(point, {
1179
+ ...defaults,
1180
+ ...options,
1181
+ includeLifecycleEvidence,
1182
+ lifecycleEvents: includeLifecycleEvidence === false ? void 0 : options.lifecycleEvents ?? defaults.lifecycleEvents ?? events
1183
+ });
1184
+ if (report.input) inputs.push(report.input);
1185
+ diagnostics.push(...report.diagnostics);
1186
+ }
1187
+ return {
1188
+ inputs,
1189
+ diagnostics
1190
+ };
1191
+ },
1192
+ clear: () => {
1193
+ decisions.length = 0;
1194
+ events.length = 0;
1195
+ }
1196
+ };
1223
1197
  }
1224
1198
  function resolveDecisionKind(point, override, diagnostics) {
1225
- const kind = override ?? point.kind;
1226
- if (isBeliefDecisionKind(kind)) return kind;
1227
- diagnostics.push({
1228
- decisionId: point.id,
1229
- severity: "error",
1230
- reason: `unsupported decisionKind "${kind}"`
1231
- });
1232
- return void 0;
1199
+ const kind = override ?? point.kind;
1200
+ if (isBeliefDecisionKind(kind)) return kind;
1201
+ diagnostics.push({
1202
+ decisionId: point.id,
1203
+ severity: "error",
1204
+ reason: `unsupported decisionKind "${kind}"`
1205
+ });
1233
1206
  }
1234
1207
  function runtimeEvidenceToBeliefEvidence(ref, point) {
1235
- if (isBeliefEvidenceSource(ref.source)) {
1236
- return {
1237
- source: ref.source,
1238
- id: ref.id,
1239
- runId: point.runId,
1240
- detail: ref.detail,
1241
- quality: ref.quality,
1242
- metadata: ref.metadata
1243
- };
1244
- }
1245
- return {
1246
- source: "event",
1247
- id: ref.id,
1248
- runId: point.runId,
1249
- detail: ref.detail,
1250
- quality: ref.quality,
1251
- metadata: mergeMetadata({ runtimeSource: ref.source }, ref.metadata)
1252
- };
1208
+ if (isBeliefEvidenceSource(ref.source)) return {
1209
+ source: ref.source,
1210
+ id: ref.id,
1211
+ runId: point.runId,
1212
+ detail: ref.detail,
1213
+ quality: ref.quality,
1214
+ metadata: ref.metadata
1215
+ };
1216
+ return {
1217
+ source: "event",
1218
+ id: ref.id,
1219
+ runId: point.runId,
1220
+ detail: ref.detail,
1221
+ quality: ref.quality,
1222
+ metadata: mergeMetadata({ runtimeSource: ref.source }, ref.metadata)
1223
+ };
1253
1224
  }
1254
1225
  function runtimeHookEventsToEvidenceRefs(point, options) {
1255
- if (options.includeLifecycleEvidence === false) return [];
1256
- return (options.lifecycleEvents ?? []).filter((event) => runtimeHookEventMatchesDecision(point, event)).map(runtimeHookEventToEvidenceRef);
1226
+ if (options.includeLifecycleEvidence === false) return [];
1227
+ return (options.lifecycleEvents ?? []).filter((event) => runtimeHookEventMatchesDecision(point, event)).map(runtimeHookEventToEvidenceRef);
1257
1228
  }
1258
1229
  function runtimeHookEventMatchesDecision(point, event) {
1259
- if (event.runId !== point.runId) return false;
1260
- if (event.scenarioId && point.scenarioId && event.scenarioId !== point.scenarioId) return false;
1261
- return event.stepIndex === void 0 || event.stepIndex === point.stepIndex;
1230
+ if (event.runId !== point.runId) return false;
1231
+ if (event.scenarioId && point.scenarioId && event.scenarioId !== point.scenarioId) return false;
1232
+ return event.stepIndex === void 0 || event.stepIndex === point.stepIndex;
1262
1233
  }
1263
1234
  function runtimeHookEventToEvidenceRef(event) {
1264
- return {
1265
- source: "runtime_event",
1266
- id: event.id,
1267
- detail: `${event.target}:${event.phase}`,
1268
- quality: "direct",
1269
- metadata: mergeMetadata(
1270
- compactMetadata({
1271
- target: event.target,
1272
- phase: event.phase,
1273
- timestamp: event.timestamp,
1274
- stepIndex: event.stepIndex,
1275
- parentId: event.parentId,
1276
- payloadPreview: previewUnknown(event.payload)
1277
- }),
1278
- event.metadata
1279
- )
1280
- };
1235
+ return {
1236
+ source: "runtime_event",
1237
+ id: event.id,
1238
+ detail: `${event.target}:${event.phase}`,
1239
+ quality: "direct",
1240
+ metadata: mergeMetadata(compactMetadata$1({
1241
+ target: event.target,
1242
+ phase: event.phase,
1243
+ timestamp: event.timestamp,
1244
+ stepIndex: event.stepIndex,
1245
+ parentId: event.parentId,
1246
+ payloadPreview: previewUnknown(event.payload)
1247
+ }), event.metadata)
1248
+ };
1281
1249
  }
1282
1250
  function lifecycleMetadata(refs) {
1283
- if (refs.length === 0) return void 0;
1284
- return {
1285
- lifecycleEventCount: refs.length,
1286
- lifecycleEventIds: refs.map((ref) => ref.id)
1287
- };
1251
+ if (refs.length === 0) return void 0;
1252
+ return {
1253
+ lifecycleEventCount: refs.length,
1254
+ lifecycleEventIds: refs.map((ref) => ref.id)
1255
+ };
1288
1256
  }
1289
1257
  function snapshotRuntimeHookEvent(event) {
1290
- return {
1291
- id: event.id,
1292
- runId: event.runId,
1293
- scenarioId: event.scenarioId,
1294
- target: event.target,
1295
- phase: event.phase,
1296
- timestamp: event.timestamp,
1297
- stepIndex: event.stepIndex,
1298
- parentId: event.parentId,
1299
- payload: snapshotUnknown(event.payload),
1300
- metadata: event.metadata ? { ...event.metadata } : void 0
1301
- };
1258
+ return {
1259
+ id: event.id,
1260
+ runId: event.runId,
1261
+ scenarioId: event.scenarioId,
1262
+ target: event.target,
1263
+ phase: event.phase,
1264
+ timestamp: event.timestamp,
1265
+ stepIndex: event.stepIndex,
1266
+ parentId: event.parentId,
1267
+ payload: snapshotUnknown(event.payload),
1268
+ metadata: event.metadata ? { ...event.metadata } : void 0
1269
+ };
1302
1270
  }
1303
1271
  function snapshotRuntimeDecisionPoint(point) {
1304
- return {
1305
- id: point.id,
1306
- runId: point.runId,
1307
- scenarioId: point.scenarioId,
1308
- stepIndex: point.stepIndex,
1309
- kind: point.kind,
1310
- candidateActions: [...point.candidateActions ?? []],
1311
- context: point.context,
1312
- evidence: (point.evidence ?? []).map((ref) => ({
1313
- source: ref.source,
1314
- id: ref.id,
1315
- detail: ref.detail,
1316
- quality: ref.quality,
1317
- metadata: ref.metadata ? { ...ref.metadata } : void 0
1318
- })),
1319
- metadata: point.metadata ? { ...point.metadata } : void 0
1320
- };
1272
+ return {
1273
+ id: point.id,
1274
+ runId: point.runId,
1275
+ scenarioId: point.scenarioId,
1276
+ stepIndex: point.stepIndex,
1277
+ kind: point.kind,
1278
+ candidateActions: [...point.candidateActions ?? []],
1279
+ context: point.context,
1280
+ evidence: (point.evidence ?? []).map((ref) => ({
1281
+ source: ref.source,
1282
+ id: ref.id,
1283
+ detail: ref.detail,
1284
+ quality: ref.quality,
1285
+ metadata: ref.metadata ? { ...ref.metadata } : void 0
1286
+ })),
1287
+ metadata: point.metadata ? { ...point.metadata } : void 0
1288
+ };
1321
1289
  }
1322
1290
  function mergeMetadata(base, extra) {
1323
- if (!base && !extra) return void 0;
1324
- return { ...base ?? {}, ...extra ?? {} };
1291
+ if (!base && !extra) return void 0;
1292
+ return {
1293
+ ...base ?? {},
1294
+ ...extra ?? {}
1295
+ };
1325
1296
  }
1326
- function compactMetadata(values) {
1327
- const entries = Object.entries(values).filter(([, value]) => value !== void 0);
1328
- return entries.length > 0 ? Object.fromEntries(entries) : void 0;
1297
+ function compactMetadata$1(values) {
1298
+ const entries = Object.entries(values).filter(([, value]) => value !== void 0);
1299
+ return entries.length > 0 ? Object.fromEntries(entries) : void 0;
1329
1300
  }
1330
1301
  function previewUnknown(value, maxChars = DEFAULT_PAYLOAD_PREVIEW_CHARS) {
1331
- if (value === void 0) return void 0;
1332
- if (typeof value === "string") return trimText(value, maxChars);
1333
- try {
1334
- return trimText(JSON.stringify(value), maxChars);
1335
- } catch {
1336
- return trimText(String(value), maxChars);
1337
- }
1302
+ if (value === void 0) return void 0;
1303
+ if (typeof value === "string") return trimText$1(value, maxChars);
1304
+ try {
1305
+ return trimText$1(JSON.stringify(value), maxChars);
1306
+ } catch {
1307
+ return trimText$1(String(value), maxChars);
1308
+ }
1338
1309
  }
1339
1310
  function snapshotUnknown(value) {
1340
- if (Array.isArray(value)) return [...value];
1341
- if (isRecord2(value)) return { ...value };
1342
- return value;
1311
+ if (Array.isArray(value)) return [...value];
1312
+ if (isRecord$1(value)) return { ...value };
1313
+ return value;
1343
1314
  }
1344
- function isRecord2(value) {
1345
- return typeof value === "object" && value !== null && !Array.isArray(value);
1315
+ function isRecord$1(value) {
1316
+ return typeof value === "object" && value !== null && !Array.isArray(value);
1346
1317
  }
1347
- function uniqueStrings(values) {
1348
- return [...new Set(values.filter((value) => value.length > 0))];
1318
+ function uniqueStrings$1(values) {
1319
+ return [...new Set(values.filter((value) => value.length > 0))];
1349
1320
  }
1350
- function trimText(value, maxChars = DEFAULT_MAX_CONTEXT_CHARS) {
1351
- if (!value) return void 0;
1352
- return value.length > maxChars ? value.slice(value.length - maxChars) : value;
1321
+ function trimText$1(value, maxChars = DEFAULT_MAX_CONTEXT_CHARS$1) {
1322
+ if (!value) return void 0;
1323
+ return value.length > maxChars ? value.slice(value.length - maxChars) : value;
1353
1324
  }
1354
1325
  function stringOrUndefined(value) {
1355
- return typeof value === "string" && value.length > 0 ? value : void 0;
1326
+ return typeof value === "string" && value.length > 0 ? value : void 0;
1356
1327
  }
1357
1328
  function finiteNumberOrUndefined(value) {
1358
- return typeof value === "number" && Number.isFinite(value) ? value : void 0;
1329
+ return typeof value === "number" && Number.isFinite(value) ? value : void 0;
1359
1330
  }
1360
1331
  function unitProbabilityOrUndefined(value) {
1361
- const number = finiteNumberOrUndefined(value);
1362
- return number !== void 0 && number >= 0 && number <= 1 ? number : void 0;
1332
+ const number = finiteNumberOrUndefined(value);
1333
+ return number !== void 0 && number >= 0 && number <= 1 ? number : void 0;
1363
1334
  }
1364
1335
  function nonNegativeNumberOrUndefined(value) {
1365
- const number = finiteNumberOrUndefined(value);
1366
- return number !== void 0 && number >= 0 ? number : void 0;
1336
+ const number = finiteNumberOrUndefined(value);
1337
+ return number !== void 0 && number >= 0 ? number : void 0;
1367
1338
  }
1368
-
1369
- // src/belief-state/phase0-measurement.ts
1370
- var DEFAULT_BASELINE_POLICY_ID = "always-accept-observed-action";
1339
+ //#endregion
1340
+ //#region src/belief-state/phase0-measurement.ts
1341
+ const DEFAULT_BASELINE_POLICY_ID = "always-accept-observed-action";
1371
1342
  function buildRuntimeBeliefPhase0Measurement(options) {
1372
- const runsById = new Map(options.runs.map((run) => [run.runId, run]));
1373
- const labelsByDecisionId = /* @__PURE__ */ new Map();
1374
- const diagnostics = [];
1375
- for (const label of options.labels) {
1376
- if (labelsByDecisionId.has(label.decisionId)) {
1377
- diagnostics.push(`${label.decisionId}: duplicate label; using the last label`);
1378
- }
1379
- labelsByDecisionId.set(label.decisionId, label);
1380
- }
1381
- const points = [];
1382
- let missingRunRecordCount = 0;
1383
- let missingLabelCount = 0;
1384
- for (const decision of options.decisions) {
1385
- const run = runsById.get(decision.runId);
1386
- if (!run) {
1387
- missingRunRecordCount += 1;
1388
- diagnostics.push(`${decision.id}: missing RunRecord join for runId ${decision.runId}`);
1389
- continue;
1390
- }
1391
- const label = labelsByDecisionId.get(decision.id);
1392
- if (!label) {
1393
- missingLabelCount += 1;
1394
- diagnostics.push(`${decision.id}: missing observed action/outcome label`);
1395
- continue;
1396
- }
1397
- const splitTag = label.splitTag ?? run.splitTag;
1398
- const report = runtimeDecisionPointToBeliefDecisionPoint(
1399
- { ...decision, scenarioId: decision.scenarioId ?? run.scenarioId },
1400
- {
1401
- chosenAction: label.chosenAction,
1402
- confidence: label.confidence,
1403
- behaviorProb: label.behaviorProb,
1404
- targetProb: label.targetProb,
1405
- qHatChosen: label.qHatChosen,
1406
- vHatTarget: label.vHatTarget,
1407
- costUsd: label.costUsd,
1408
- outcome: label.outcome,
1409
- lifecycleEvents: options.events,
1410
- metadata: compactMetadata2({
1411
- baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
1412
- splitTag,
1413
- ...label.metadata
1414
- })
1415
- }
1416
- );
1417
- diagnostics.push(...report.diagnostics.map((item) => `${item.decisionId}: ${item.reason}`));
1418
- if (report.point) points.push(report.point);
1419
- }
1420
- const packet = buildBeliefDecisionResearchEvidencePacket({
1421
- ...options,
1422
- points
1423
- });
1424
- return {
1425
- points,
1426
- packet,
1427
- summary: summarizePhase0Measurement(options, points, packet, {
1428
- missingRunRecordCount,
1429
- missingLabelCount
1430
- }),
1431
- diagnostics
1432
- };
1343
+ const runsById = new Map(options.runs.map((run) => [run.runId, run]));
1344
+ const labelsByDecisionId = /* @__PURE__ */ new Map();
1345
+ const diagnostics = [];
1346
+ for (const label of options.labels) {
1347
+ if (labelsByDecisionId.has(label.decisionId)) diagnostics.push(`${label.decisionId}: duplicate label; using the last label`);
1348
+ labelsByDecisionId.set(label.decisionId, label);
1349
+ }
1350
+ const points = [];
1351
+ let missingRunRecordCount = 0;
1352
+ let missingLabelCount = 0;
1353
+ for (const decision of options.decisions) {
1354
+ const run = runsById.get(decision.runId);
1355
+ if (!run) {
1356
+ missingRunRecordCount += 1;
1357
+ diagnostics.push(`${decision.id}: missing RunRecord join for runId ${decision.runId}`);
1358
+ continue;
1359
+ }
1360
+ const label = labelsByDecisionId.get(decision.id);
1361
+ if (!label) {
1362
+ missingLabelCount += 1;
1363
+ diagnostics.push(`${decision.id}: missing observed action/outcome label`);
1364
+ continue;
1365
+ }
1366
+ const splitTag = label.splitTag ?? run.splitTag;
1367
+ const report = runtimeDecisionPointToBeliefDecisionPoint({
1368
+ ...decision,
1369
+ scenarioId: decision.scenarioId ?? run.scenarioId
1370
+ }, {
1371
+ chosenAction: label.chosenAction,
1372
+ confidence: label.confidence,
1373
+ behaviorProb: label.behaviorProb,
1374
+ targetProb: label.targetProb,
1375
+ qHatChosen: label.qHatChosen,
1376
+ vHatTarget: label.vHatTarget,
1377
+ costUsd: label.costUsd,
1378
+ outcome: label.outcome,
1379
+ lifecycleEvents: options.events,
1380
+ metadata: compactMetadata({
1381
+ baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
1382
+ splitTag,
1383
+ ...label.metadata
1384
+ })
1385
+ });
1386
+ diagnostics.push(...report.diagnostics.map((item) => `${item.decisionId}: ${item.reason}`));
1387
+ if (report.point) points.push(report.point);
1388
+ }
1389
+ const packet = buildBeliefDecisionResearchEvidencePacket({
1390
+ ...options,
1391
+ points
1392
+ });
1393
+ return {
1394
+ points,
1395
+ packet,
1396
+ summary: summarizePhase0Measurement(options, points, packet, {
1397
+ missingRunRecordCount,
1398
+ missingLabelCount
1399
+ }),
1400
+ diagnostics
1401
+ };
1433
1402
  }
1434
1403
  function summarizePhase0Measurement(options, points, packet, counts) {
1435
- const producerDecisionCount = options.decisions.length;
1436
- return {
1437
- runCount: options.runs.length,
1438
- producerDecisionCount,
1439
- lifecycleEventCount: options.events?.length ?? 0,
1440
- labelCount: options.labels.length,
1441
- completedPointCount: points.length,
1442
- runJoinRate: ratio(producerDecisionCount - counts.missingRunRecordCount, producerDecisionCount),
1443
- labelJoinRate: ratio(points.length, producerDecisionCount),
1444
- missingRunRecordCount: counts.missingRunRecordCount,
1445
- missingLabelCount: counts.missingLabelCount,
1446
- withEvidence: points.filter((point) => point.evidence.length > 0).length,
1447
- withOutcome: points.filter((point) => point.outcome).length,
1448
- withSplit: points.filter((point) => typeof point.metadata?.splitTag === "string").length,
1449
- withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
1450
- withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
1451
- baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
1452
- packetStatus: packet.status,
1453
- claimScope: packet.claimScope
1454
- };
1404
+ const producerDecisionCount = options.decisions.length;
1405
+ return {
1406
+ runCount: options.runs.length,
1407
+ producerDecisionCount,
1408
+ lifecycleEventCount: options.events?.length ?? 0,
1409
+ labelCount: options.labels.length,
1410
+ completedPointCount: points.length,
1411
+ runJoinRate: ratio(producerDecisionCount - counts.missingRunRecordCount, producerDecisionCount),
1412
+ labelJoinRate: ratio(points.length, producerDecisionCount),
1413
+ missingRunRecordCount: counts.missingRunRecordCount,
1414
+ missingLabelCount: counts.missingLabelCount,
1415
+ withEvidence: points.filter((point) => point.evidence.length > 0).length,
1416
+ withOutcome: points.filter((point) => point.outcome).length,
1417
+ withSplit: points.filter((point) => typeof point.metadata?.splitTag === "string").length,
1418
+ withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
1419
+ withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
1420
+ baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
1421
+ packetStatus: packet.status,
1422
+ claimScope: packet.claimScope
1423
+ };
1455
1424
  }
1456
1425
  function ratio(numerator, denominator) {
1457
- return denominator > 0 ? numerator / denominator : 0;
1426
+ return denominator > 0 ? numerator / denominator : 0;
1458
1427
  }
1459
- function compactMetadata2(values) {
1460
- const entries = Object.entries(values).filter(([, value]) => value !== void 0);
1461
- return entries.length > 0 ? Object.fromEntries(entries) : void 0;
1462
- }
1463
-
1464
- // src/belief-state/runtime-benchmark-corpus.ts
1465
- var MAX_STRING_LENGTH = 12e3;
1466
- var MAX_CONTEXT_LENGTH = 2e4;
1467
- var MAX_EVIDENCE_DETAIL_LENGTH = 2e3;
1468
- var MAX_CANDIDATE_ACTIONS = 50;
1469
- var MAX_EVIDENCE_REFS = 50;
1470
- var MAX_METADATA_DEPTH = 4;
1471
- var MAX_METADATA_KEYS = 100;
1472
- var SENSITIVE_KEY_RE = /(?:authorization|api[_-]?key|token|secret|password|cookie|credential|bearer)/i;
1473
- var SENSITIVE_VALUE_RES = [
1474
- /\bBearer\s+[A-Za-z0-9._~+/=-]+/gi,
1475
- /\b(?:sk|gh[pousr])_[A-Za-z0-9_]{20,}\b/g,
1476
- /\b(?:sk|ghp|gho|ghu|ghs|ghr)-[A-Za-z0-9_-]{20,}\b/g
1428
+ function compactMetadata(values) {
1429
+ const entries = Object.entries(values).filter(([, value]) => value !== void 0);
1430
+ return entries.length > 0 ? Object.fromEntries(entries) : void 0;
1431
+ }
1432
+ //#endregion
1433
+ //#region src/belief-state/runtime-benchmark-corpus.ts
1434
+ const MAX_STRING_LENGTH = 12e3;
1435
+ const MAX_CONTEXT_LENGTH = 2e4;
1436
+ const MAX_EVIDENCE_DETAIL_LENGTH = 2e3;
1437
+ const MAX_CANDIDATE_ACTIONS = 50;
1438
+ const MAX_EVIDENCE_REFS = 50;
1439
+ const MAX_METADATA_DEPTH = 4;
1440
+ const MAX_METADATA_KEYS = 100;
1441
+ const SENSITIVE_KEY_RE = /(?:authorization|api[_-]?key|token|secret|password|cookie|credential|bearer)/i;
1442
+ const SENSITIVE_VALUE_RES = [
1443
+ /\bBearer\s+[A-Za-z0-9._~+/=-]+/gi,
1444
+ /\b(?:sk|gh[pousr])_[A-Za-z0-9_]{20,}\b/g,
1445
+ /\b(?:sk|ghp|gho|ghu|ghs|ghr)-[A-Za-z0-9_-]{20,}\b/g
1477
1446
  ];
1478
- var SENSITIVE_ASSIGNMENT_RE = /\b(api[_-]?key|token|secret|password|cookie)\s*[:=]\s*["']?[^"'\s,;}]+/gi;
1447
+ const SENSITIVE_ASSIGNMENT_RE = /\b(api[_-]?key|token|secret|password|cookie)\s*[:=]\s*["']?[^"'\s,;}]+/gi;
1479
1448
  function buildRuntimeBenchmarkBeliefPhase0Measurement(options) {
1480
- const diagnostics = [];
1481
- const trajectory = projectRuntimeTrajectoryEvidence({
1482
- records: options.records,
1483
- defaultSplitTag: options.defaultSplitTag,
1484
- recordIdOf: runtimeBenchmarkRecordId,
1485
- scenarioIdOf: runtimeBenchmarkScenarioId
1486
- });
1487
- const decisions = options.decisions ?? runtimeBenchmarkDecisionPoints(options.records, diagnostics);
1488
- const labels = options.labels ?? [];
1489
- if (decisions.length === 0) {
1490
- diagnostics.push(
1491
- "no runtime decision points supplied or found on records; benchmark lifecycle events alone cannot produce belief decision rows"
1492
- );
1493
- }
1494
- if (labels.length === 0 && decisions.length > 0) {
1495
- diagnostics.push(
1496
- "no decision labels supplied; observed action/outcome joins will be incomplete"
1497
- );
1498
- }
1499
- const measurement = buildRuntimeBeliefPhase0Measurement({
1500
- ...options,
1501
- runs: trajectory.runs,
1502
- events: trajectory.events,
1503
- decisions,
1504
- labels
1505
- });
1506
- return {
1507
- runs: trajectory.runs,
1508
- events: trajectory.events,
1509
- decisions,
1510
- labels,
1511
- trajectory,
1512
- measurement,
1513
- summary: {
1514
- decisionCount: decisions.length,
1515
- labelCount: labels.length
1516
- },
1517
- diagnostics: [...trajectory.diagnostics, ...diagnostics, ...measurement.diagnostics]
1518
- };
1449
+ const diagnostics = [];
1450
+ const trajectory = projectRuntimeTrajectoryEvidence({
1451
+ records: options.records,
1452
+ defaultSplitTag: options.defaultSplitTag,
1453
+ recordIdOf: runtimeBenchmarkRecordId,
1454
+ scenarioIdOf: runtimeBenchmarkScenarioId
1455
+ });
1456
+ const decisions = options.decisions ?? runtimeBenchmarkDecisionPoints(options.records, diagnostics);
1457
+ const labels = options.labels ?? [];
1458
+ if (decisions.length === 0) diagnostics.push("no runtime decision points supplied or found on records; benchmark lifecycle events alone cannot produce belief decision rows");
1459
+ if (labels.length === 0 && decisions.length > 0) diagnostics.push("no decision labels supplied; observed action/outcome joins will be incomplete");
1460
+ const measurement = buildRuntimeBeliefPhase0Measurement({
1461
+ ...options,
1462
+ runs: trajectory.runs,
1463
+ events: trajectory.events,
1464
+ decisions,
1465
+ labels
1466
+ });
1467
+ return {
1468
+ runs: trajectory.runs,
1469
+ events: trajectory.events,
1470
+ decisions,
1471
+ labels,
1472
+ trajectory,
1473
+ measurement,
1474
+ summary: {
1475
+ decisionCount: decisions.length,
1476
+ labelCount: labels.length
1477
+ },
1478
+ diagnostics: [
1479
+ ...trajectory.diagnostics,
1480
+ ...diagnostics,
1481
+ ...measurement.diagnostics
1482
+ ]
1483
+ };
1519
1484
  }
1520
1485
  function runtimeBenchmarkRecordId(record) {
1521
- const parts = [
1522
- nonEmptyString(record.benchmark),
1523
- nonEmptyString(record.instanceId),
1524
- nonEmptyString(record.condition)
1525
- ].filter((part) => part !== void 0);
1526
- return parts.length > 0 ? parts.join(":") : void 0;
1486
+ const parts = [
1487
+ nonEmptyString(record.benchmark),
1488
+ nonEmptyString(record.instanceId),
1489
+ nonEmptyString(record.condition)
1490
+ ].filter((part) => part !== void 0);
1491
+ return parts.length > 0 ? parts.join(":") : void 0;
1527
1492
  }
1528
1493
  function runtimeBenchmarkScenarioId(record) {
1529
- return nonEmptyString(record.instanceId);
1494
+ return nonEmptyString(record.instanceId);
1530
1495
  }
1531
1496
  function runtimeBenchmarkDecisionPoints(records, diagnostics) {
1532
- const decisions = [];
1533
- for (let recordIndex = 0; recordIndex < records.length; recordIndex += 1) {
1534
- const record = records[recordIndex];
1535
- const raw = record.runtimeDecisionPoints;
1536
- if (raw === void 0) continue;
1537
- const recordId = runtimeBenchmarkRecordId(record) ?? `record[${recordIndex}]`;
1538
- if (!Array.isArray(raw)) {
1539
- diagnostics.push(`${recordId}: runtimeDecisionPoints is not an array`);
1540
- continue;
1541
- }
1542
- for (let pointIndex = 0; pointIndex < raw.length; pointIndex += 1) {
1543
- const point = runtimeBenchmarkDecisionPoint(raw[pointIndex], {
1544
- diagnostics,
1545
- path: `${recordId}: runtimeDecisionPoints[${pointIndex}]`
1546
- });
1547
- if (!point) {
1548
- diagnostics.push(
1549
- `${recordId}: runtimeDecisionPoints[${pointIndex}] is not a RuntimeDecisionPoint`
1550
- );
1551
- continue;
1552
- }
1553
- decisions.push(point);
1554
- }
1555
- }
1556
- return decisions;
1497
+ const decisions = [];
1498
+ for (let recordIndex = 0; recordIndex < records.length; recordIndex += 1) {
1499
+ const record = records[recordIndex];
1500
+ const raw = record.runtimeDecisionPoints;
1501
+ if (raw === void 0) continue;
1502
+ const recordId = runtimeBenchmarkRecordId(record) ?? `record[${recordIndex}]`;
1503
+ if (!Array.isArray(raw)) {
1504
+ diagnostics.push(`${recordId}: runtimeDecisionPoints is not an array`);
1505
+ continue;
1506
+ }
1507
+ for (let pointIndex = 0; pointIndex < raw.length; pointIndex += 1) {
1508
+ const point = runtimeBenchmarkDecisionPoint(raw[pointIndex], {
1509
+ diagnostics,
1510
+ path: `${recordId}: runtimeDecisionPoints[${pointIndex}]`
1511
+ });
1512
+ if (!point) {
1513
+ diagnostics.push(`${recordId}: runtimeDecisionPoints[${pointIndex}] is not a RuntimeDecisionPoint`);
1514
+ continue;
1515
+ }
1516
+ decisions.push(point);
1517
+ }
1518
+ }
1519
+ return decisions;
1557
1520
  }
1558
1521
  function runtimeBenchmarkDecisionPoint(input, context) {
1559
- if (!isRecord3(input)) return null;
1560
- if (typeof input.id !== "string" || input.id.length === 0) return null;
1561
- if (typeof input.runId !== "string" || input.runId.length === 0) return null;
1562
- if (typeof input.stepIndex !== "number" || !Number.isInteger(input.stepIndex) || input.stepIndex < 0) {
1563
- return null;
1564
- }
1565
- if (typeof input.kind !== "string" || input.kind.length === 0) return null;
1566
- return {
1567
- id: sanitizeString(input.id, MAX_STRING_LENGTH),
1568
- runId: sanitizeString(input.runId, MAX_STRING_LENGTH),
1569
- scenarioId: sanitizeOptionalString(input.scenarioId, MAX_STRING_LENGTH),
1570
- stepIndex: input.stepIndex,
1571
- kind: sanitizeString(input.kind, MAX_STRING_LENGTH),
1572
- candidateActions: stringArray(input.candidateActions, {
1573
- ...context,
1574
- maxItems: MAX_CANDIDATE_ACTIONS,
1575
- label: "candidateActions"
1576
- }),
1577
- context: sanitizeOptionalString(input.context, MAX_CONTEXT_LENGTH),
1578
- evidence: runtimeBenchmarkEvidence(input.evidence, context),
1579
- metadata: sanitizeMetadataRecord(input.metadata)
1580
- };
1522
+ if (!isRecord(input)) return null;
1523
+ if (typeof input.id !== "string" || input.id.length === 0) return null;
1524
+ if (typeof input.runId !== "string" || input.runId.length === 0) return null;
1525
+ if (typeof input.stepIndex !== "number" || !Number.isInteger(input.stepIndex) || input.stepIndex < 0) return null;
1526
+ if (typeof input.kind !== "string" || input.kind.length === 0) return null;
1527
+ return {
1528
+ id: sanitizeString(input.id, MAX_STRING_LENGTH),
1529
+ runId: sanitizeString(input.runId, MAX_STRING_LENGTH),
1530
+ scenarioId: sanitizeOptionalString(input.scenarioId, MAX_STRING_LENGTH),
1531
+ stepIndex: input.stepIndex,
1532
+ kind: sanitizeString(input.kind, MAX_STRING_LENGTH),
1533
+ candidateActions: stringArray(input.candidateActions, {
1534
+ ...context,
1535
+ maxItems: MAX_CANDIDATE_ACTIONS,
1536
+ label: "candidateActions"
1537
+ }),
1538
+ context: sanitizeOptionalString(input.context, MAX_CONTEXT_LENGTH),
1539
+ evidence: runtimeBenchmarkEvidence(input.evidence, context),
1540
+ metadata: sanitizeMetadataRecord(input.metadata)
1541
+ };
1581
1542
  }
1582
1543
  function runtimeBenchmarkEvidence(input, context) {
1583
- if (!Array.isArray(input)) return [];
1584
- if (input.length > MAX_EVIDENCE_REFS) {
1585
- context.diagnostics.push(`${context.path}: evidence truncated to ${MAX_EVIDENCE_REFS} refs`);
1586
- }
1587
- return input.slice(0, MAX_EVIDENCE_REFS).flatMap((item) => {
1588
- if (!isRecord3(item)) return [];
1589
- const source = sanitizeOptionalString(item.source, MAX_STRING_LENGTH);
1590
- const id = sanitizeOptionalString(item.id, MAX_STRING_LENGTH);
1591
- if (!source || !id) return [];
1592
- return [
1593
- {
1594
- source,
1595
- id,
1596
- detail: sanitizeOptionalString(item.detail, MAX_EVIDENCE_DETAIL_LENGTH),
1597
- metadata: sanitizeMetadataRecord(item.metadata)
1598
- }
1599
- ];
1600
- });
1544
+ if (!Array.isArray(input)) return [];
1545
+ if (input.length > MAX_EVIDENCE_REFS) context.diagnostics.push(`${context.path}: evidence truncated to ${MAX_EVIDENCE_REFS} refs`);
1546
+ return input.slice(0, MAX_EVIDENCE_REFS).flatMap((item) => {
1547
+ if (!isRecord(item)) return [];
1548
+ const source = sanitizeOptionalString(item.source, MAX_STRING_LENGTH);
1549
+ const id = sanitizeOptionalString(item.id, MAX_STRING_LENGTH);
1550
+ if (!source || !id) return [];
1551
+ return [{
1552
+ source,
1553
+ id,
1554
+ detail: sanitizeOptionalString(item.detail, MAX_EVIDENCE_DETAIL_LENGTH),
1555
+ metadata: sanitizeMetadataRecord(item.metadata)
1556
+ }];
1557
+ });
1601
1558
  }
1602
1559
  function stringArray(input, context) {
1603
- if (!Array.isArray(input)) return void 0;
1604
- if (input.length > context.maxItems) {
1605
- context.diagnostics.push(`${context.path}: ${context.label} truncated to ${context.maxItems}`);
1606
- }
1607
- const values = input.slice(0, context.maxItems).filter((value) => typeof value === "string" && value.length > 0).map((value) => sanitizeString(value, MAX_STRING_LENGTH));
1608
- return values.length > 0 ? values : void 0;
1560
+ if (!Array.isArray(input)) return void 0;
1561
+ if (input.length > context.maxItems) context.diagnostics.push(`${context.path}: ${context.label} truncated to ${context.maxItems}`);
1562
+ const values = input.slice(0, context.maxItems).filter((value) => typeof value === "string" && value.length > 0).map((value) => sanitizeString(value, MAX_STRING_LENGTH));
1563
+ return values.length > 0 ? values : void 0;
1609
1564
  }
1610
1565
  function sanitizeMetadataRecord(metadata) {
1611
- if (!isRecord3(metadata)) return void 0;
1612
- const sanitized = sanitizeMetadata(metadata);
1613
- if (!sanitized || typeof sanitized !== "object" || Array.isArray(sanitized)) return void 0;
1614
- return sanitized;
1566
+ if (!isRecord(metadata)) return void 0;
1567
+ const sanitized = sanitizeMetadata(metadata);
1568
+ if (!sanitized || typeof sanitized !== "object" || Array.isArray(sanitized)) return void 0;
1569
+ return sanitized;
1615
1570
  }
1616
1571
  function sanitizeMetadata(value, depth = 0) {
1617
- if (value == null) return value;
1618
- if (typeof value === "string") return sanitizeString(value, MAX_STRING_LENGTH);
1619
- if (typeof value === "number" || typeof value === "boolean") return value;
1620
- if (Array.isArray(value)) {
1621
- if (depth >= MAX_METADATA_DEPTH) return "[MaxDepth]";
1622
- return value.slice(0, MAX_METADATA_KEYS).map((item) => sanitizeMetadata(item, depth + 1));
1623
- }
1624
- if (!isRecord3(value)) return void 0;
1625
- if (depth >= MAX_METADATA_DEPTH) return "[MaxDepth]";
1626
- const sanitized = {};
1627
- for (const [key, nested] of Object.entries(value).slice(0, MAX_METADATA_KEYS)) {
1628
- sanitized[key] = SENSITIVE_KEY_RE.test(key) ? "[REDACTED]" : sanitizeMetadata(nested, depth + 1);
1629
- }
1630
- return sanitized;
1572
+ if (value == null) return value;
1573
+ if (typeof value === "string") return sanitizeString(value, MAX_STRING_LENGTH);
1574
+ if (typeof value === "number" || typeof value === "boolean") return value;
1575
+ if (Array.isArray(value)) {
1576
+ if (depth >= MAX_METADATA_DEPTH) return "[MaxDepth]";
1577
+ return value.slice(0, MAX_METADATA_KEYS).map((item) => sanitizeMetadata(item, depth + 1));
1578
+ }
1579
+ if (!isRecord(value)) return void 0;
1580
+ if (depth >= MAX_METADATA_DEPTH) return "[MaxDepth]";
1581
+ const sanitized = {};
1582
+ for (const [key, nested] of Object.entries(value).slice(0, MAX_METADATA_KEYS)) sanitized[key] = SENSITIVE_KEY_RE.test(key) ? "[REDACTED]" : sanitizeMetadata(nested, depth + 1);
1583
+ return sanitized;
1631
1584
  }
1632
1585
  function sanitizeOptionalString(value, maxLength) {
1633
- return typeof value === "string" && value.length > 0 ? sanitizeString(value, maxLength) : void 0;
1586
+ return typeof value === "string" && value.length > 0 ? sanitizeString(value, maxLength) : void 0;
1634
1587
  }
1635
1588
  function sanitizeString(value, maxLength) {
1636
- let sanitized = value;
1637
- for (const pattern of SENSITIVE_VALUE_RES) {
1638
- sanitized = sanitized.replace(pattern, "[REDACTED]");
1639
- }
1640
- sanitized = sanitized.replace(
1641
- SENSITIVE_ASSIGNMENT_RE,
1642
- (_match, key) => `${key}=[REDACTED]`
1643
- );
1644
- if (sanitized.length <= maxLength) return sanitized;
1645
- return sanitized.slice(0, maxLength);
1646
- }
1647
- function isRecord3(value) {
1648
- return typeof value === "object" && value !== null && !Array.isArray(value);
1589
+ let sanitized = value;
1590
+ for (const pattern of SENSITIVE_VALUE_RES) sanitized = sanitized.replace(pattern, "[REDACTED]");
1591
+ sanitized = sanitized.replace(SENSITIVE_ASSIGNMENT_RE, (_match, key) => `${key}=[REDACTED]`);
1592
+ if (sanitized.length <= maxLength) return sanitized;
1593
+ return sanitized.slice(0, maxLength);
1594
+ }
1595
+ function isRecord(value) {
1596
+ return typeof value === "object" && value !== null && !Array.isArray(value);
1649
1597
  }
1650
1598
  function nonEmptyString(value) {
1651
- return typeof value === "string" && value.length > 0 ? value : void 0;
1599
+ return typeof value === "string" && value.length > 0 ? value : void 0;
1652
1600
  }
1653
-
1654
- // src/belief-state/shadow-probe.ts
1655
- var DEFAULT_CONCURRENCY = 4;
1656
- var DEFAULT_MAX_CONTEXT_CHARS2 = 12e3;
1601
+ //#endregion
1602
+ //#region src/belief-state/shadow-probe.ts
1603
+ const DEFAULT_CONCURRENCY = 4;
1604
+ const DEFAULT_MAX_CONTEXT_CHARS = 12e3;
1657
1605
  async function runBeliefShadowProbe(options) {
1658
- const concurrency = boundedInteger(options.concurrency ?? DEFAULT_CONCURRENCY, 1, 32);
1659
- const records = [];
1660
- const diagnostics = [];
1661
- let next = 0;
1662
- async function worker() {
1663
- while (next < options.points.length) {
1664
- const index = next;
1665
- next += 1;
1666
- const point = options.points[index];
1667
- if (!point) continue;
1668
- const result = await probePoint(point, options);
1669
- records[index] = result.record;
1670
- diagnostics.push(...result.diagnostics);
1671
- }
1672
- }
1673
- await Promise.all(Array.from({ length: Math.min(concurrency, options.points.length) }, worker));
1674
- const completed = records.filter((record) => !!record);
1675
- return {
1676
- probeId: options.probeId,
1677
- records: completed,
1678
- diagnostics,
1679
- summary: summarizeShadowProbe(options.points.length, completed)
1680
- };
1606
+ const concurrency = boundedInteger(options.concurrency ?? DEFAULT_CONCURRENCY, 1, 32);
1607
+ const records = [];
1608
+ const diagnostics = [];
1609
+ let next = 0;
1610
+ async function worker() {
1611
+ while (next < options.points.length) {
1612
+ const index = next;
1613
+ next += 1;
1614
+ const point = options.points[index];
1615
+ if (!point) continue;
1616
+ const result = await probePoint(point, options);
1617
+ records[index] = result.record;
1618
+ diagnostics.push(...result.diagnostics);
1619
+ }
1620
+ }
1621
+ await Promise.all(Array.from({ length: Math.min(concurrency, options.points.length) }, worker));
1622
+ const completed = records.filter((record) => !!record);
1623
+ return {
1624
+ probeId: options.probeId,
1625
+ records: completed,
1626
+ diagnostics,
1627
+ summary: summarizeShadowProbe(options.points.length, completed)
1628
+ };
1681
1629
  }
1682
1630
  function formatBeliefShadowProbePrompt(input) {
1683
- return [
1684
- "Return only JSON. Do not include chain-of-thought.",
1685
- "Infer the agent belief state at this decision boundary using only the context below.",
1686
- "",
1687
- `decisionKind: ${input.decisionKind}`,
1688
- `candidateActions: ${JSON.stringify(input.candidateActions)}`,
1689
- input.observedAction ? `observedAction: ${JSON.stringify(input.observedAction)}` : "",
1690
- input.context ? `context:
1691
- ${input.context}` : "",
1692
- "",
1693
- "Schema:",
1694
- JSON.stringify({
1695
- predictedAction: "one candidate action",
1696
- confidence: "number in [0,1]",
1697
- beliefSummary: "short outcome-blind summary",
1698
- uncertainty: ["short uncertainty"],
1699
- evidenceRefs: ["evidence id"],
1700
- wouldChangeMindIf: ["observable evidence"],
1701
- targetProb: "optional number in [0,1]",
1702
- qHatChosen: "optional number in [0,1], paired with vHatTarget",
1703
- vHatTarget: "optional number in [0,1], paired with qHatChosen"
1704
- })
1705
- ].filter(Boolean).join("\n");
1631
+ return [
1632
+ "Return only JSON. Do not include chain-of-thought.",
1633
+ "Infer the agent belief state at this decision boundary using only the context below.",
1634
+ "",
1635
+ `decisionKind: ${input.decisionKind}`,
1636
+ `candidateActions: ${JSON.stringify(input.candidateActions)}`,
1637
+ input.observedAction ? `observedAction: ${JSON.stringify(input.observedAction)}` : "",
1638
+ input.context ? `context:\n${input.context}` : "",
1639
+ "",
1640
+ "Schema:",
1641
+ JSON.stringify({
1642
+ predictedAction: "one candidate action",
1643
+ confidence: "number in [0,1]",
1644
+ beliefSummary: "short outcome-blind summary",
1645
+ uncertainty: ["short uncertainty"],
1646
+ evidenceRefs: ["evidence id"],
1647
+ wouldChangeMindIf: ["observable evidence"],
1648
+ targetProb: "optional number in [0,1]",
1649
+ qHatChosen: "optional number in [0,1], paired with vHatTarget",
1650
+ vHatTarget: "optional number in [0,1], paired with qHatChosen"
1651
+ })
1652
+ ].filter(Boolean).join("\n");
1706
1653
  }
1707
1654
  async function probePoint(point, options) {
1708
- const diagnostics = [];
1709
- const candidateActions = uniqueStrings2(point.candidateActions ?? []);
1710
- if ((options.requireCandidateActions ?? true) && candidateActions.length === 0) {
1711
- diagnostics.push({
1712
- decisionId: point.id,
1713
- severity: "warning",
1714
- reason: "missing candidateActions"
1715
- });
1716
- return { diagnostics };
1717
- }
1718
- let response;
1719
- try {
1720
- response = await options.probe({
1721
- probeId: options.probeId,
1722
- decisionId: point.id,
1723
- runId: point.runId,
1724
- scenarioId: point.scenarioId,
1725
- stepIndex: point.stepIndex,
1726
- decisionKind: point.kind,
1727
- candidateActions,
1728
- ...options.includeObservedAction ? { observedAction: point.chosenAction } : {},
1729
- evidence: point.evidence.map((ref) => ({
1730
- id: ref.id,
1731
- source: ref.source,
1732
- ...options.includeEvidenceDetail && ref.detail ? { detail: ref.detail } : {},
1733
- ...ref.quality ? { quality: ref.quality } : {}
1734
- })),
1735
- context: trimText2(await options.contextOf?.(point), options.maxContextChars),
1736
- metadata: await options.metadataOf?.(point)
1737
- });
1738
- } catch (error) {
1739
- diagnostics.push({
1740
- decisionId: point.id,
1741
- severity: "error",
1742
- reason: `probe threw: ${errorMessage2(error)}`
1743
- });
1744
- return { diagnostics };
1745
- }
1746
- const normalized = normalizeProbeResponse(response, {
1747
- point,
1748
- candidateActions,
1749
- allowOutOfSetActions: options.allowOutOfSetActions ?? false
1750
- });
1751
- if (!normalized.record) {
1752
- diagnostics.push(...normalized.diagnostics);
1753
- return { diagnostics };
1754
- }
1755
- return {
1756
- record: {
1757
- probeId: options.probeId,
1758
- decisionId: point.id,
1759
- runId: point.runId,
1760
- scenarioId: point.scenarioId,
1761
- stepIndex: point.stepIndex,
1762
- decisionKind: point.kind,
1763
- candidateActions,
1764
- observedAction: point.chosenAction,
1765
- agreesWithObservedAction: normalized.record.predictedAction === point.chosenAction,
1766
- ...options.includeOutcomeInRecord === false ? {} : { outcome: point.outcome },
1767
- ...normalized.record
1768
- },
1769
- diagnostics
1770
- };
1655
+ const diagnostics = [];
1656
+ const candidateActions = uniqueStrings(point.candidateActions ?? []);
1657
+ if ((options.requireCandidateActions ?? true) && candidateActions.length === 0) {
1658
+ diagnostics.push({
1659
+ decisionId: point.id,
1660
+ severity: "warning",
1661
+ reason: "missing candidateActions"
1662
+ });
1663
+ return { diagnostics };
1664
+ }
1665
+ let response;
1666
+ try {
1667
+ response = await options.probe({
1668
+ probeId: options.probeId,
1669
+ decisionId: point.id,
1670
+ runId: point.runId,
1671
+ scenarioId: point.scenarioId,
1672
+ stepIndex: point.stepIndex,
1673
+ decisionKind: point.kind,
1674
+ candidateActions,
1675
+ ...options.includeObservedAction ? { observedAction: point.chosenAction } : {},
1676
+ evidence: point.evidence.map((ref) => ({
1677
+ id: ref.id,
1678
+ source: ref.source,
1679
+ ...options.includeEvidenceDetail && ref.detail ? { detail: ref.detail } : {},
1680
+ ...ref.quality ? { quality: ref.quality } : {}
1681
+ })),
1682
+ context: trimText(await options.contextOf?.(point), options.maxContextChars),
1683
+ metadata: await options.metadataOf?.(point)
1684
+ });
1685
+ } catch (error) {
1686
+ diagnostics.push({
1687
+ decisionId: point.id,
1688
+ severity: "error",
1689
+ reason: `probe threw: ${errorMessage(error)}`
1690
+ });
1691
+ return { diagnostics };
1692
+ }
1693
+ const normalized = normalizeProbeResponse(response, {
1694
+ point,
1695
+ candidateActions,
1696
+ allowOutOfSetActions: options.allowOutOfSetActions ?? false
1697
+ });
1698
+ if (!normalized.record) {
1699
+ diagnostics.push(...normalized.diagnostics);
1700
+ return { diagnostics };
1701
+ }
1702
+ return {
1703
+ record: {
1704
+ probeId: options.probeId,
1705
+ decisionId: point.id,
1706
+ runId: point.runId,
1707
+ scenarioId: point.scenarioId,
1708
+ stepIndex: point.stepIndex,
1709
+ decisionKind: point.kind,
1710
+ candidateActions,
1711
+ observedAction: point.chosenAction,
1712
+ agreesWithObservedAction: normalized.record.predictedAction === point.chosenAction,
1713
+ ...options.includeOutcomeInRecord === false ? {} : { outcome: point.outcome },
1714
+ ...normalized.record
1715
+ },
1716
+ diagnostics
1717
+ };
1771
1718
  }
1772
1719
  function normalizeProbeResponse(response, options) {
1773
- const diagnostics = [];
1774
- const predictedAction = stringOrNull(response.predictedAction);
1775
- if (!predictedAction) {
1776
- diagnostics.push({
1777
- decisionId: options.point.id,
1778
- severity: "error",
1779
- reason: "missing predictedAction"
1780
- });
1781
- } else if (!options.allowOutOfSetActions && options.candidateActions.length > 0 && !options.candidateActions.includes(predictedAction)) {
1782
- diagnostics.push({
1783
- decisionId: options.point.id,
1784
- severity: "error",
1785
- reason: `predictedAction ${predictedAction} is not in candidateActions`
1786
- });
1787
- }
1788
- if (!isUnitProbability(response.confidence)) {
1789
- diagnostics.push({
1790
- decisionId: options.point.id,
1791
- severity: "error",
1792
- reason: `invalid confidence ${String(response.confidence)}`
1793
- });
1794
- }
1795
- if (response.targetProb !== void 0 && !isUnitProbability(response.targetProb)) {
1796
- diagnostics.push({
1797
- decisionId: options.point.id,
1798
- severity: "error",
1799
- reason: `invalid targetProb ${String(response.targetProb)}`
1800
- });
1801
- }
1802
- const hasQHatChosen = response.qHatChosen !== void 0 && response.qHatChosen !== null;
1803
- const hasVHatTarget = response.vHatTarget !== void 0 && response.vHatTarget !== null;
1804
- if (hasQHatChosen !== hasVHatTarget) {
1805
- diagnostics.push({
1806
- decisionId: options.point.id,
1807
- severity: "error",
1808
- reason: "qHatChosen and vHatTarget must be supplied together"
1809
- });
1810
- }
1811
- if (hasQHatChosen && !isUnitProbability(response.qHatChosen)) {
1812
- diagnostics.push({
1813
- decisionId: options.point.id,
1814
- severity: "error",
1815
- reason: `invalid qHatChosen ${String(response.qHatChosen)}`
1816
- });
1817
- }
1818
- if (hasVHatTarget && !isUnitProbability(response.vHatTarget)) {
1819
- diagnostics.push({
1820
- decisionId: options.point.id,
1821
- severity: "error",
1822
- reason: `invalid vHatTarget ${String(response.vHatTarget)}`
1823
- });
1824
- }
1825
- if (diagnostics.length > 0 || !predictedAction) return { diagnostics };
1826
- return {
1827
- record: {
1828
- predictedAction,
1829
- confidence: response.confidence,
1830
- ...response.beliefSummary ? { beliefSummary: trimText2(response.beliefSummary, 2e3) } : {},
1831
- uncertainty: compactStrings(response.uncertainty),
1832
- evidenceRefs: compactStrings(response.evidenceRefs),
1833
- wouldChangeMindIf: compactStrings(response.wouldChangeMindIf),
1834
- ...response.targetProb !== void 0 ? { targetProb: response.targetProb } : {},
1835
- ...response.qHatChosen !== void 0 ? { qHatChosen: response.qHatChosen } : {},
1836
- ...response.vHatTarget !== void 0 ? { vHatTarget: response.vHatTarget } : {},
1837
- ...response.metadata ? { metadata: response.metadata } : {}
1838
- },
1839
- diagnostics
1840
- };
1720
+ const diagnostics = [];
1721
+ const predictedAction = stringOrNull(response.predictedAction);
1722
+ if (!predictedAction) diagnostics.push({
1723
+ decisionId: options.point.id,
1724
+ severity: "error",
1725
+ reason: "missing predictedAction"
1726
+ });
1727
+ else if (!options.allowOutOfSetActions && options.candidateActions.length > 0 && !options.candidateActions.includes(predictedAction)) diagnostics.push({
1728
+ decisionId: options.point.id,
1729
+ severity: "error",
1730
+ reason: `predictedAction ${predictedAction} is not in candidateActions`
1731
+ });
1732
+ if (!isUnitProbability(response.confidence)) diagnostics.push({
1733
+ decisionId: options.point.id,
1734
+ severity: "error",
1735
+ reason: `invalid confidence ${String(response.confidence)}`
1736
+ });
1737
+ if (response.targetProb !== void 0 && !isUnitProbability(response.targetProb)) diagnostics.push({
1738
+ decisionId: options.point.id,
1739
+ severity: "error",
1740
+ reason: `invalid targetProb ${String(response.targetProb)}`
1741
+ });
1742
+ const hasQHatChosen = response.qHatChosen !== void 0 && response.qHatChosen !== null;
1743
+ const hasVHatTarget = response.vHatTarget !== void 0 && response.vHatTarget !== null;
1744
+ if (hasQHatChosen !== hasVHatTarget) diagnostics.push({
1745
+ decisionId: options.point.id,
1746
+ severity: "error",
1747
+ reason: "qHatChosen and vHatTarget must be supplied together"
1748
+ });
1749
+ if (hasQHatChosen && !isUnitProbability(response.qHatChosen)) diagnostics.push({
1750
+ decisionId: options.point.id,
1751
+ severity: "error",
1752
+ reason: `invalid qHatChosen ${String(response.qHatChosen)}`
1753
+ });
1754
+ if (hasVHatTarget && !isUnitProbability(response.vHatTarget)) diagnostics.push({
1755
+ decisionId: options.point.id,
1756
+ severity: "error",
1757
+ reason: `invalid vHatTarget ${String(response.vHatTarget)}`
1758
+ });
1759
+ if (diagnostics.length > 0 || !predictedAction) return { diagnostics };
1760
+ return {
1761
+ record: {
1762
+ predictedAction,
1763
+ confidence: response.confidence,
1764
+ ...response.beliefSummary ? { beliefSummary: trimText(response.beliefSummary, 2e3) } : {},
1765
+ uncertainty: compactStrings(response.uncertainty),
1766
+ evidenceRefs: compactStrings(response.evidenceRefs),
1767
+ wouldChangeMindIf: compactStrings(response.wouldChangeMindIf),
1768
+ ...response.targetProb !== void 0 ? { targetProb: response.targetProb } : {},
1769
+ ...response.qHatChosen !== void 0 ? { qHatChosen: response.qHatChosen } : {},
1770
+ ...response.vHatTarget !== void 0 ? { vHatTarget: response.vHatTarget } : {},
1771
+ ...response.metadata ? { metadata: response.metadata } : {}
1772
+ },
1773
+ diagnostics
1774
+ };
1841
1775
  }
1842
1776
  function summarizeShadowProbe(attempted, records) {
1843
- const confidences = records.map((record) => record.confidence);
1844
- const agreements = records.filter((record) => record.agreesWithObservedAction).length;
1845
- return {
1846
- attempted,
1847
- completed: records.length,
1848
- dropped: attempted - records.length,
1849
- withOutcome: records.filter((record) => record.outcome !== void 0).length,
1850
- withTargetProb: records.filter((record) => record.targetProb !== void 0).length,
1851
- meanConfidence: confidences.length > 0 ? mean3(confidences) : null,
1852
- observedAgreementRate: records.length > 0 ? agreements / records.length : null
1853
- };
1777
+ const confidences = records.map((record) => record.confidence);
1778
+ const agreements = records.filter((record) => record.agreesWithObservedAction).length;
1779
+ return {
1780
+ attempted,
1781
+ completed: records.length,
1782
+ dropped: attempted - records.length,
1783
+ withOutcome: records.filter((record) => record.outcome !== void 0).length,
1784
+ withTargetProb: records.filter((record) => record.targetProb !== void 0).length,
1785
+ meanConfidence: confidences.length > 0 ? mean(confidences) : null,
1786
+ observedAgreementRate: records.length > 0 ? agreements / records.length : null
1787
+ };
1854
1788
  }
1855
1789
  function isUnitProbability(value) {
1856
- return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
1790
+ return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
1857
1791
  }
1858
1792
  function boundedInteger(value, min, max) {
1859
- if (!Number.isFinite(value)) return min;
1860
- return Math.max(min, Math.min(max, Math.floor(value)));
1793
+ if (!Number.isFinite(value)) return min;
1794
+ return Math.max(min, Math.min(max, Math.floor(value)));
1861
1795
  }
1862
1796
  function compactStrings(values, maxItems = 12) {
1863
- if (!Array.isArray(values)) return [];
1864
- return values.filter((value) => typeof value === "string" && value.length > 0).slice(0, maxItems).map((value) => trimText2(value, 500) ?? "").filter(Boolean);
1797
+ if (!Array.isArray(values)) return [];
1798
+ return values.filter((value) => typeof value === "string" && value.length > 0).slice(0, maxItems).map((value) => trimText(value, 500) ?? "").filter(Boolean);
1865
1799
  }
1866
- function uniqueStrings2(values) {
1867
- return [...new Set(values.filter((value) => value.length > 0))];
1800
+ function uniqueStrings(values) {
1801
+ return [...new Set(values.filter((value) => value.length > 0))];
1868
1802
  }
1869
1803
  function stringOrNull(value) {
1870
- return typeof value === "string" && value.length > 0 ? value : null;
1871
- }
1872
- function trimText2(value, maxChars = DEFAULT_MAX_CONTEXT_CHARS2) {
1873
- if (!value) return void 0;
1874
- return value.length > maxChars ? value.slice(value.length - maxChars) : value;
1875
- }
1876
- function mean3(values) {
1877
- return values.reduce((sum, value) => sum + value, 0) / values.length;
1878
- }
1879
- function errorMessage2(error) {
1880
- return error instanceof Error ? error.message : String(error);
1881
- }
1882
- export {
1883
- BELIEF_DECISION_KINDS,
1884
- BELIEF_EVALUATION_CRITERIA,
1885
- BELIEF_EVIDENCE_QUALITIES,
1886
- BELIEF_EVIDENCE_SOURCES,
1887
- analyzeBeliefDecisionCorpus,
1888
- analyzeBeliefPolicy,
1889
- beliefDecisionsToOffPolicyTrajectories,
1890
- buildBeliefDecisionResearchEvidencePacket,
1891
- buildCodeAgentBeliefEvidenceCorpus,
1892
- buildRuntimeBeliefPhase0Measurement,
1893
- buildRuntimeBenchmarkBeliefPhase0Measurement,
1894
- calibrateBeliefDecisions,
1895
- createBeliefRuntimeHookCollector,
1896
- embeddedBeliefOpeTargetPolicy,
1897
- evaluateBeliefOffPolicy,
1898
- evaluateBeliefSelectivePolicy,
1899
- extractBeliefDecisionPoints,
1900
- extractCodeAgentBeliefDecisionPoints,
1901
- formatBeliefShadowProbePrompt,
1902
- inventoryBeliefDecisionPoints,
1903
- isBeliefDecisionKind,
1904
- isBeliefEvidenceSource,
1905
- runBeliefShadowProbe,
1906
- runtimeDecisionPointToBeliefDecisionPoint,
1907
- runtimeDecisionPointToBeliefShadowProbeInput,
1908
- selectBeliefDecisionTarget,
1909
- thresholdSelectivePolicy
1910
- };
1804
+ return typeof value === "string" && value.length > 0 ? value : null;
1805
+ }
1806
+ function trimText(value, maxChars = DEFAULT_MAX_CONTEXT_CHARS) {
1807
+ if (!value) return void 0;
1808
+ return value.length > maxChars ? value.slice(value.length - maxChars) : value;
1809
+ }
1810
+ function mean(values) {
1811
+ return values.reduce((sum, value) => sum + value, 0) / values.length;
1812
+ }
1813
+ function errorMessage(error) {
1814
+ return error instanceof Error ? error.message : String(error);
1815
+ }
1816
+ //#endregion
1817
+ export { BELIEF_DECISION_KINDS, BELIEF_EVALUATION_CRITERIA, BELIEF_EVIDENCE_QUALITIES, BELIEF_EVIDENCE_SOURCES, analyzeBeliefDecisionCorpus, analyzeBeliefPolicy, beliefDecisionsToOffPolicyTrajectories, buildBeliefDecisionResearchEvidencePacket, buildCodeAgentBeliefEvidenceCorpus, buildRuntimeBeliefPhase0Measurement, buildRuntimeBenchmarkBeliefPhase0Measurement, calibrateBeliefDecisions, createBeliefRuntimeHookCollector, embeddedBeliefOpeTargetPolicy, evaluateBeliefOffPolicy, evaluateBeliefSelectivePolicy, extractBeliefDecisionPoints, extractCodeAgentBeliefDecisionPoints, formatBeliefShadowProbePrompt, inventoryBeliefDecisionPoints, isBeliefDecisionKind, isBeliefEvidenceSource, runBeliefShadowProbe, runtimeDecisionPointToBeliefDecisionPoint, runtimeDecisionPointToBeliefShadowProbeInput, selectBeliefDecisionTarget, thresholdSelectivePolicy };
1818
+
1911
1819
  //# sourceMappingURL=index.js.map