@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
@@ -1,1912 +1,1819 @@
1
- import {
2
- fromClaudeCodeSession,
3
- fromCodexSession,
4
- fromKimiCodeSession,
5
- fromOpenCodeSession,
6
- fromPiSession,
7
- observeCodeAgentSession
8
- } from "../chunk-SZLVEKMJ.js";
9
- import {
10
- calibrationFromPairs
11
- } from "../chunk-NPCTHQIO.js";
12
- import {
13
- projectRuntimeTrajectoryEvidence
14
- } from "../chunk-T4SQEITX.js";
15
- import {
16
- offPolicyEstimateAll
17
- } from "../chunk-VGRCHJON.js";
18
- import {
19
- confidenceInterval
20
- } from "../chunk-MHELPNRP.js";
21
- import "../chunk-VI2UW6B6.js";
22
- import "../chunk-PXE2VKMX.js";
23
- import {
24
- ValidationError
25
- } from "../chunk-ONWEPEDO.js";
26
- import "../chunk-PZ5AY32C.js";
27
-
28
- // src/belief-state/calibration.ts
1
+ import { s as ValidationError } from "../errors-8YnH8WlF.js";
2
+ import { a as confidenceInterval } from "../statistics-CnnxdpOg.js";
3
+ import { s as trainingScore } from "../reward-nw2xZGZG.js";
4
+ import { n as projectRuntimeTrajectoryEvidence } from "../runtime-trajectory-1gyaTOoC.js";
5
+ import { r as offPolicyEstimateAll } from "../off-policy-DvgzvtIx.js";
6
+ import { n as calibrationFromPairs } from "../calibration-CNWWA6K8.js";
7
+ import { a as fromPiSession, c as observeCodeAgentSession, i as fromOpenCodeSession, n as fromCodexSession, r as fromKimiCodeSession, t as fromClaudeCodeSession } from "../code-agent-session-BjkMTQ7H.js";
8
+ //#region src/belief-state/calibration.ts
29
9
  function calibrateBeliefDecisions(points, options = {}) {
30
- const filtered = filterCalibrationRegion(points, options);
31
- const pairs = filtered.filter((point) => typeof point.confidence === "number" && point.outcome).map((point) => ({
32
- evalScore: point.confidence,
33
- outcome: outcomeScore(point)
34
- })).filter((pair) => Number.isFinite(pair.outcome));
35
- const minPairs = options.minPairs ?? 10;
36
- if (pairs.length < minPairs) return null;
37
- return calibrationFromPairs(pairs, "belief-confidence", "decision-outcome", {
38
- bins: options.bins ?? 5,
39
- range: { lo: 0, hi: 1 }
40
- });
10
+ const pairs = filterCalibrationRegion(points, options).filter((point) => typeof point.confidence === "number" && point.outcome).map((point) => ({
11
+ evalScore: point.confidence,
12
+ outcome: outcomeScore$1(point)
13
+ })).filter((pair) => Number.isFinite(pair.outcome));
14
+ const minPairs = options.minPairs ?? 10;
15
+ if (pairs.length < minPairs) return null;
16
+ return calibrationFromPairs(pairs, "belief-confidence", "decision-outcome", {
17
+ bins: options.bins ?? 5,
18
+ range: {
19
+ lo: 0,
20
+ hi: 1
21
+ }
22
+ });
41
23
  }
42
24
  function filterCalibrationRegion(points, options) {
43
- const region = options.region ?? "all";
44
- if (region === "all") return points;
45
- const policy = options.policy;
46
- if (!policy) {
47
- throw new ValidationError(
48
- `calibrateBeliefDecisions: policy is required when region is "${region}"`
49
- );
50
- }
51
- return points.filter((point) => {
52
- const accepted = policy.decide(point).action === "accept";
53
- return region === "accepted" ? accepted : !accepted;
54
- });
55
- }
56
- function outcomeScore(point) {
57
- if (typeof point.outcome?.reward === "number") return point.outcome.reward;
58
- if (typeof point.outcome?.score === "number") return point.outcome.score;
59
- if (point.outcome?.success === true) return 1;
60
- if (point.outcome?.success === false) return 0;
61
- return Number.NaN;
62
- }
63
-
64
- // src/belief-state/ope.ts
25
+ const region = options.region ?? "all";
26
+ if (region === "all") return points;
27
+ const policy = options.policy;
28
+ if (!policy) throw new ValidationError(`calibrateBeliefDecisions: policy is required when region is "${region}"`);
29
+ return points.filter((point) => {
30
+ const accepted = policy.decide(point).action === "accept";
31
+ return region === "accepted" ? accepted : !accepted;
32
+ });
33
+ }
34
+ function outcomeScore$1(point) {
35
+ if (typeof point.outcome?.reward === "number") return point.outcome.reward;
36
+ if (typeof point.outcome?.score === "number") return point.outcome.score;
37
+ if (point.outcome?.success === true) return 1;
38
+ if (point.outcome?.success === false) return 0;
39
+ return NaN;
40
+ }
41
+ //#endregion
42
+ //#region src/belief-state/ope.ts
65
43
  function embeddedBeliefOpeTargetPolicy(id = "embedded-target-prob") {
66
- return {
67
- id,
68
- targetProbOf(point) {
69
- return point.targetProb;
70
- },
71
- qHatChosenOf(point) {
72
- return point.qHatChosen;
73
- },
74
- vHatTargetOf(point) {
75
- return point.vHatTarget;
76
- },
77
- qHatOf(point) {
78
- return point.qHat;
79
- }
80
- };
44
+ return {
45
+ id,
46
+ targetProbOf(point) {
47
+ return point.targetProb;
48
+ },
49
+ qHatChosenOf(point) {
50
+ return point.qHatChosen;
51
+ },
52
+ vHatTargetOf(point) {
53
+ return point.vHatTarget;
54
+ }
55
+ };
81
56
  }
82
57
  function beliefDecisionsToOffPolicyTrajectories(points, targetPolicy, options = {}) {
83
- const trajectories = [];
84
- const diagnostics = [];
85
- for (const point of points) {
86
- if (!point.outcome) {
87
- diagnostics.push(`${point.id}: missing outcome`);
88
- continue;
89
- }
90
- if (!isBehaviorProbability(point.behaviorProb)) {
91
- diagnostics.push(`${point.id}: invalid behaviorProb ${formatProbability(point.behaviorProb)}`);
92
- continue;
93
- }
94
- let targetProb;
95
- let qHatChosen;
96
- let vHatTarget;
97
- let qHat;
98
- try {
99
- targetProb = targetPolicy.targetProbOf(point);
100
- qHatChosen = targetPolicy.qHatChosenOf?.(point);
101
- vHatTarget = targetPolicy.vHatTargetOf?.(point);
102
- qHat = targetPolicy.qHatOf?.(point);
103
- } catch (error) {
104
- diagnostics.push(
105
- `${point.id}: target policy ${targetPolicy.id} threw (${errorMessage(error)})`
106
- );
107
- continue;
108
- }
109
- if (!isTargetProbability(targetProb)) {
110
- diagnostics.push(`${point.id}: invalid targetProb ${formatProbability(targetProb)}`);
111
- continue;
112
- }
113
- const hasQHatChosen = qHatChosen !== null && qHatChosen !== void 0;
114
- const hasVHatTarget = vHatTarget !== null && vHatTarget !== void 0;
115
- if (hasQHatChosen !== hasVHatTarget) {
116
- diagnostics.push(`${point.id}: qHatChosen and vHatTarget must be supplied together`);
117
- continue;
118
- }
119
- if (hasQHatChosen && hasVHatTarget && (!isTargetProbability(qHatChosen) || !isTargetProbability(vHatTarget))) {
120
- diagnostics.push(
121
- `${point.id}: invalid contextual Q pair qHatChosen=${formatProbability(qHatChosen)} vHatTarget=${formatProbability(vHatTarget)}`
122
- );
123
- continue;
124
- }
125
- if (!hasQHatChosen && !hasVHatTarget && qHat !== null && qHat !== void 0 && !isTargetProbability(qHat)) {
126
- diagnostics.push(`${point.id}: invalid qHat ${formatProbability(qHat)}; ignoring qHat`);
127
- qHat = null;
128
- }
129
- trajectories.push({
130
- runId: point.id,
131
- reward: rewardOf(point),
132
- behaviorProb: point.behaviorProb,
133
- targetProb,
134
- ...qHatChosen !== void 0 ? { qHatChosen } : {},
135
- ...vHatTarget !== void 0 ? { vHatTarget } : {},
136
- qHat
137
- });
138
- }
139
- return {
140
- targetPolicyId: targetPolicy.id,
141
- trajectories,
142
- dropped: points.length - trajectories.length,
143
- diagnostics: compactDiagnostics(diagnostics, options.maxDiagnostics ?? 20)
144
- };
58
+ const trajectories = [];
59
+ const diagnostics = [];
60
+ for (const point of points) {
61
+ if (!point.outcome) {
62
+ diagnostics.push(`${point.id}: missing outcome`);
63
+ continue;
64
+ }
65
+ if (!isBehaviorProbability(point.behaviorProb)) {
66
+ diagnostics.push(`${point.id}: invalid behaviorProb ${formatProbability(point.behaviorProb)}`);
67
+ continue;
68
+ }
69
+ let targetProb;
70
+ let qHatChosen;
71
+ let vHatTarget;
72
+ try {
73
+ targetProb = targetPolicy.targetProbOf(point);
74
+ qHatChosen = targetPolicy.qHatChosenOf?.(point);
75
+ vHatTarget = targetPolicy.vHatTargetOf?.(point);
76
+ } catch (error) {
77
+ diagnostics.push(`${point.id}: target policy ${targetPolicy.id} threw (${errorMessage$1(error)})`);
78
+ continue;
79
+ }
80
+ if (!isTargetProbability(targetProb)) {
81
+ diagnostics.push(`${point.id}: invalid targetProb ${formatProbability(targetProb)}`);
82
+ continue;
83
+ }
84
+ const hasQHatChosen = qHatChosen !== null && qHatChosen !== void 0;
85
+ const hasVHatTarget = vHatTarget !== null && vHatTarget !== void 0;
86
+ if (hasQHatChosen !== hasVHatTarget) {
87
+ diagnostics.push(`${point.id}: qHatChosen and vHatTarget must be supplied together`);
88
+ continue;
89
+ }
90
+ if (hasQHatChosen && hasVHatTarget && (!isTargetProbability(qHatChosen) || !isTargetProbability(vHatTarget))) {
91
+ diagnostics.push(`${point.id}: invalid contextual Q pair qHatChosen=${formatProbability(qHatChosen)} vHatTarget=${formatProbability(vHatTarget)}`);
92
+ continue;
93
+ }
94
+ trajectories.push({
95
+ runId: point.id,
96
+ reward: rewardOf$1(point),
97
+ behaviorProb: point.behaviorProb,
98
+ targetProb,
99
+ ...qHatChosen !== void 0 ? { qHatChosen } : {},
100
+ ...vHatTarget !== void 0 ? { vHatTarget } : {}
101
+ });
102
+ }
103
+ return {
104
+ targetPolicyId: targetPolicy.id,
105
+ trajectories,
106
+ dropped: points.length - trajectories.length,
107
+ diagnostics: compactDiagnostics(diagnostics, options.maxDiagnostics ?? 20)
108
+ };
145
109
  }
146
110
  function evaluateBeliefOffPolicy(points, targetPolicy, options = {}) {
147
- const trajectoryReport = beliefDecisionsToOffPolicyTrajectories(points, targetPolicy, options);
148
- const { trajectories } = trajectoryReport;
149
- const estimates = offPolicyEstimateAll(trajectories, options);
150
- const support = supportDiagnostics(estimates.dr, {
151
- minEffectiveSampleSize: options.minEffectiveSampleSize ?? 30,
152
- minEffectiveSampleRatio: options.minEffectiveSampleRatio ?? 0.25,
153
- dropped: trajectoryReport.dropped,
154
- diagnostics: trajectoryReport.diagnostics,
155
- legacyScalarContributions: estimates.dr.contributionCounts?.legacyScalar ?? 0
156
- });
157
- return { targetPolicyId: targetPolicy.id, ...estimates, support };
111
+ const trajectoryReport = beliefDecisionsToOffPolicyTrajectories(points, targetPolicy, options);
112
+ const { trajectories } = trajectoryReport;
113
+ const estimates = offPolicyEstimateAll(trajectories, options);
114
+ const support = supportDiagnostics(estimates.dr, {
115
+ minEffectiveSampleSize: options.minEffectiveSampleSize ?? 30,
116
+ minEffectiveSampleRatio: options.minEffectiveSampleRatio ?? .25,
117
+ dropped: trajectoryReport.dropped,
118
+ diagnostics: trajectoryReport.diagnostics
119
+ });
120
+ return {
121
+ targetPolicyId: targetPolicy.id,
122
+ ...estimates,
123
+ support
124
+ };
158
125
  }
159
126
  function supportDiagnostics(estimate, options) {
160
- const ratio2 = estimate.n > 0 ? estimate.effectiveSampleSize / estimate.n : 0;
161
- const reasons = [...options.diagnostics];
162
- if (estimate.n === 0) {
163
- reasons.push("no valid OPE trajectories");
164
- }
165
- if (options.dropped > 0) {
166
- reasons.push(`dropped ${options.dropped} unsupported decision(s)`);
167
- }
168
- if (options.legacyScalarContributions > 0) {
169
- reasons.push(
170
- `${options.legacyScalarContributions} decision(s) used deprecated scalar qHat; supply qHatChosen and vHatTarget for contextual doubly robust estimation`
171
- );
172
- }
173
- if (estimate.effectiveSampleSize < options.minEffectiveSampleSize) {
174
- reasons.push(
175
- `effective sample size ${estimate.effectiveSampleSize.toFixed(2)} below ${options.minEffectiveSampleSize}`
176
- );
177
- }
178
- if (ratio2 < options.minEffectiveSampleRatio) {
179
- reasons.push(
180
- `effective sample ratio ${ratio2.toFixed(2)} below ${options.minEffectiveSampleRatio}`
181
- );
182
- }
183
- if (estimate.maxImportanceWeight > 10) {
184
- reasons.push(`max importance weight ${estimate.maxImportanceWeight.toFixed(2)} is high`);
185
- }
186
- return {
187
- supported: reasons.length === 0,
188
- n: estimate.n,
189
- dropped: options.dropped,
190
- effectiveSampleSize: estimate.effectiveSampleSize,
191
- effectiveSampleRatio: ratio2,
192
- maxImportanceWeight: estimate.maxImportanceWeight,
193
- reasons
194
- };
195
- }
196
- function rewardOf(point) {
197
- if (typeof point.outcome?.reward === "number") return point.outcome.reward;
198
- if (typeof point.outcome?.score === "number") return point.outcome.score;
199
- if (point.outcome?.success === true) return 1;
200
- return 0;
127
+ const ratio = estimate.n > 0 ? estimate.effectiveSampleSize / estimate.n : 0;
128
+ const reasons = [...options.diagnostics];
129
+ if (estimate.n === 0) reasons.push("no valid OPE trajectories");
130
+ if (options.dropped > 0) reasons.push(`dropped ${options.dropped} unsupported decision(s)`);
131
+ if (estimate.effectiveSampleSize < options.minEffectiveSampleSize) reasons.push(`effective sample size ${estimate.effectiveSampleSize.toFixed(2)} below ${options.minEffectiveSampleSize}`);
132
+ if (ratio < options.minEffectiveSampleRatio) reasons.push(`effective sample ratio ${ratio.toFixed(2)} below ${options.minEffectiveSampleRatio}`);
133
+ if (estimate.maxImportanceWeight > 10) reasons.push(`max importance weight ${estimate.maxImportanceWeight.toFixed(2)} is high`);
134
+ return {
135
+ supported: reasons.length === 0,
136
+ n: estimate.n,
137
+ dropped: options.dropped,
138
+ effectiveSampleSize: estimate.effectiveSampleSize,
139
+ effectiveSampleRatio: ratio,
140
+ maxImportanceWeight: estimate.maxImportanceWeight,
141
+ reasons
142
+ };
143
+ }
144
+ function rewardOf$1(point) {
145
+ if (typeof point.outcome?.reward === "number") return point.outcome.reward;
146
+ if (typeof point.outcome?.score === "number") return point.outcome.score;
147
+ if (point.outcome?.success === true) return 1;
148
+ return 0;
201
149
  }
202
150
  function isBehaviorProbability(value) {
203
- return typeof value === "number" && Number.isFinite(value) && value > 0 && value <= 1;
151
+ return typeof value === "number" && Number.isFinite(value) && value > 0 && value <= 1;
204
152
  }
205
153
  function isTargetProbability(value) {
206
- return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
154
+ return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
207
155
  }
208
156
  function formatProbability(value) {
209
- return typeof value === "number" ? String(value) : String(value ?? "missing");
157
+ return typeof value === "number" ? String(value) : String(value ?? "missing");
210
158
  }
211
- function errorMessage(error) {
212
- return error instanceof Error ? error.message : String(error);
159
+ function errorMessage$1(error) {
160
+ return error instanceof Error ? error.message : String(error);
213
161
  }
214
162
  function compactDiagnostics(diagnostics, maxDiagnostics) {
215
- if (diagnostics.length <= maxDiagnostics) return diagnostics;
216
- return [
217
- ...diagnostics.slice(0, maxDiagnostics),
218
- `${diagnostics.length - maxDiagnostics} additional OPE diagnostic(s) omitted`
219
- ];
220
- }
221
-
222
- // src/belief-state/selective.ts
223
- var DEFAULT_UTILITY = {
224
- successUtility: 1,
225
- failureUtility: -1,
226
- deferUtility: 0,
227
- verifyCost: 0.05,
228
- askCost: 0.05,
229
- retryCost: 0.1,
230
- stopUtility: 0,
231
- costWeight: 1
163
+ if (diagnostics.length <= maxDiagnostics) return diagnostics;
164
+ return [...diagnostics.slice(0, maxDiagnostics), `${diagnostics.length - maxDiagnostics} additional OPE diagnostic(s) omitted`];
165
+ }
166
+ //#endregion
167
+ //#region src/belief-state/selective.ts
168
+ const DEFAULT_UTILITY = {
169
+ successUtility: 1,
170
+ failureUtility: -1,
171
+ deferUtility: 0,
172
+ verifyCost: .05,
173
+ askCost: .05,
174
+ retryCost: .1,
175
+ stopUtility: 0,
176
+ costWeight: 1
232
177
  };
233
178
  function thresholdSelectivePolicy(options) {
234
- const threshold = options.confidenceThreshold;
235
- if (!Number.isFinite(threshold) || threshold < 0 || threshold > 1) {
236
- throw new ValidationError(
237
- `thresholdSelectivePolicy: confidenceThreshold must be in [0, 1], got ${threshold}`
238
- );
239
- }
240
- const belowThresholdAction = options.belowThresholdAction ?? "verify";
241
- return {
242
- id: options.id ?? `confidence>=${threshold}`,
243
- decide(point) {
244
- const confidence = point.confidence ?? 0;
245
- return {
246
- action: confidence >= threshold ? "accept" : belowThresholdAction,
247
- confidence,
248
- targetProb: point.targetProb,
249
- qHatChosen: point.qHatChosen,
250
- vHatTarget: point.vHatTarget,
251
- qHat: point.qHat,
252
- reason: confidence >= threshold ? "confidence threshold passed" : "confidence threshold failed"
253
- };
254
- }
255
- };
179
+ const threshold = options.confidenceThreshold;
180
+ if (!Number.isFinite(threshold) || threshold < 0 || threshold > 1) throw new ValidationError(`thresholdSelectivePolicy: confidenceThreshold must be in [0, 1], got ${threshold}`);
181
+ const belowThresholdAction = options.belowThresholdAction ?? "verify";
182
+ return {
183
+ id: options.id ?? `confidence>=${threshold}`,
184
+ decide(point) {
185
+ const confidence = point.confidence ?? 0;
186
+ return {
187
+ action: confidence >= threshold ? "accept" : belowThresholdAction,
188
+ confidence,
189
+ targetProb: point.targetProb,
190
+ qHatChosen: point.qHatChosen,
191
+ vHatTarget: point.vHatTarget,
192
+ reason: confidence >= threshold ? "confidence threshold passed" : "confidence threshold failed"
193
+ };
194
+ }
195
+ };
256
196
  }
257
197
  function evaluateBeliefSelectivePolicy(points, policy, options = {}) {
258
- const utility = { ...DEFAULT_UTILITY, ...options.utility ?? {} };
259
- const scored = points.filter((point) => point.outcome);
260
- const minN = options.minN ?? 30;
261
- const minAccepted = options.minAccepted ?? 5;
262
- const minUtilityDelta = options.minUtilityDelta ?? 0;
263
- const deltas = [];
264
- const acceptedRewards = [];
265
- const rejectedRewards = [];
266
- let baselineUtility = 0;
267
- let policyUtility = 0;
268
- let accepted = 0;
269
- let acceptedErrors = 0;
270
- for (const point of scored) {
271
- const baseline = acceptUtility(point, utility);
272
- const decision = policy.decide(point);
273
- const candidate = policyDecisionUtility(point, decision.action, utility);
274
- const reward = rewardOf2(point, utility);
275
- baselineUtility += baseline;
276
- policyUtility += candidate;
277
- deltas.push(candidate - baseline);
278
- if (decision.action === "accept") {
279
- accepted++;
280
- acceptedRewards.push(reward);
281
- if (reward < 0) acceptedErrors++;
282
- } else {
283
- rejectedRewards.push(reward);
284
- }
285
- }
286
- const n = scored.length;
287
- const rejected = Math.max(0, n - accepted);
288
- const ci = confidenceInterval(deltas, 0.95, { seed: options.seed ?? 17 });
289
- const reasons = [];
290
- if (n < minN) reasons.push(`need at least ${minN} scored decisions, got ${n}`);
291
- if (accepted < minAccepted)
292
- reasons.push(`need at least ${minAccepted} accepted decisions, got ${accepted}`);
293
- if (ci.lower <= minUtilityDelta) {
294
- reasons.push(`utility CI lower bound ${ci.lower.toFixed(4)} does not clear ${minUtilityDelta}`);
295
- }
296
- const recommendation = n < minN || accepted < minAccepted ? "need_more_data" : ci.lower > minUtilityDelta ? "ship" : "hold";
297
- return {
298
- policyId: policy.id,
299
- n,
300
- accepted,
301
- rejected,
302
- coverage: n > 0 ? accepted / n : 0,
303
- acceptedErrorRate: accepted > 0 ? acceptedErrors / accepted : 0,
304
- baselineUtility,
305
- policyUtility,
306
- utilityDelta: policyUtility - baselineUtility,
307
- utilityCi95: ci,
308
- rejectedMeanReward: rejectedRewards.length > 0 ? mean(rejectedRewards) : null,
309
- recommendation,
310
- reasons
311
- };
198
+ const utility = {
199
+ ...DEFAULT_UTILITY,
200
+ ...options.utility ?? {}
201
+ };
202
+ const scored = points.filter((point) => point.outcome);
203
+ const minN = options.minN ?? 30;
204
+ const minAccepted = options.minAccepted ?? 5;
205
+ const minUtilityDelta = options.minUtilityDelta ?? 0;
206
+ const deltas = [];
207
+ const acceptedRewards = [];
208
+ const rejectedRewards = [];
209
+ let baselineUtility = 0;
210
+ let policyUtility = 0;
211
+ let accepted = 0;
212
+ let acceptedErrors = 0;
213
+ for (const point of scored) {
214
+ const baseline = acceptUtility(point, utility);
215
+ const decision = policy.decide(point);
216
+ const candidate = policyDecisionUtility(point, decision.action, utility);
217
+ const reward = rewardOf(point, utility);
218
+ baselineUtility += baseline;
219
+ policyUtility += candidate;
220
+ deltas.push(candidate - baseline);
221
+ if (decision.action === "accept") {
222
+ accepted++;
223
+ acceptedRewards.push(reward);
224
+ if (reward < 0) acceptedErrors++;
225
+ } else rejectedRewards.push(reward);
226
+ }
227
+ const n = scored.length;
228
+ const rejected = Math.max(0, n - accepted);
229
+ const ci = confidenceInterval(deltas, .95, { seed: options.seed ?? 17 });
230
+ const reasons = [];
231
+ if (n < minN) reasons.push(`need at least ${minN} scored decisions, got ${n}`);
232
+ if (accepted < minAccepted) reasons.push(`need at least ${minAccepted} accepted decisions, got ${accepted}`);
233
+ if (ci.lower <= minUtilityDelta) reasons.push(`utility CI lower bound ${ci.lower.toFixed(4)} does not clear ${minUtilityDelta}`);
234
+ const recommendation = n < minN || accepted < minAccepted ? "need_more_data" : ci.lower > minUtilityDelta ? "ship" : "hold";
235
+ return {
236
+ policyId: policy.id,
237
+ n,
238
+ accepted,
239
+ rejected,
240
+ coverage: n > 0 ? accepted / n : 0,
241
+ acceptedErrorRate: accepted > 0 ? acceptedErrors / accepted : 0,
242
+ baselineUtility,
243
+ policyUtility,
244
+ utilityDelta: policyUtility - baselineUtility,
245
+ utilityCi95: ci,
246
+ rejectedMeanReward: rejectedRewards.length > 0 ? mean$2(rejectedRewards) : null,
247
+ recommendation,
248
+ reasons
249
+ };
312
250
  }
313
251
  function acceptUtility(point, utility) {
314
- return rewardOf2(point, utility) - utility.costWeight * (point.costUsd ?? point.outcome?.costUsd ?? 0);
252
+ return rewardOf(point, utility) - utility.costWeight * (point.costUsd ?? point.outcome?.costUsd ?? 0);
315
253
  }
316
254
  function policyDecisionUtility(point, action, utility) {
317
- if (action === "accept") return acceptUtility(point, utility);
318
- if (action === "verify") return utility.deferUtility - utility.verifyCost;
319
- if (action === "ask") return utility.deferUtility - utility.askCost;
320
- if (action === "retry") return utility.deferUtility - utility.retryCost;
321
- if (action === "stop") return utility.stopUtility;
322
- return utility.deferUtility;
323
- }
324
- function rewardOf2(point, utility) {
325
- const outcome = point.outcome;
326
- if (!outcome) return utility.failureUtility;
327
- if (typeof outcome.reward === "number") return 2 * outcome.reward - 1;
328
- if (typeof outcome.score === "number") return 2 * outcome.score - 1;
329
- if (outcome.success === true) return utility.successUtility;
330
- if (outcome.success === false) return utility.failureUtility;
331
- return utility.failureUtility;
332
- }
333
- function mean(values) {
334
- return values.reduce((sum, value) => sum + value, 0) / values.length;
335
- }
336
-
337
- // src/belief-state/report.ts
255
+ if (action === "accept") return acceptUtility(point, utility);
256
+ if (action === "verify") return utility.deferUtility - utility.verifyCost;
257
+ if (action === "ask") return utility.deferUtility - utility.askCost;
258
+ if (action === "retry") return utility.deferUtility - utility.retryCost;
259
+ if (action === "stop") return utility.stopUtility;
260
+ return utility.deferUtility;
261
+ }
262
+ function rewardOf(point, utility) {
263
+ const outcome = point.outcome;
264
+ if (!outcome) return utility.failureUtility;
265
+ if (typeof outcome.reward === "number") return 2 * outcome.reward - 1;
266
+ if (typeof outcome.score === "number") return 2 * outcome.score - 1;
267
+ if (outcome.success === true) return utility.successUtility;
268
+ if (outcome.success === false) return utility.failureUtility;
269
+ return utility.failureUtility;
270
+ }
271
+ function mean$2(values) {
272
+ return values.reduce((sum, value) => sum + value, 0) / values.length;
273
+ }
274
+ //#endregion
275
+ //#region src/belief-state/report.ts
338
276
  function analyzeBeliefPolicy(options) {
339
- const selective = evaluateBeliefSelectivePolicy(options.points, options.policy, options.selective);
340
- const calibration = calibrateBeliefDecisions(options.points, options.calibration);
341
- const opeTargetPolicy = options.ope?.targetPolicy;
342
- const ope = opeTargetPolicy ? evaluateBeliefOffPolicy(options.points, opeTargetPolicy, options.ope) : null;
343
- const diagnostics = [];
344
- const selectiveStatus = selective.recommendation;
345
- const calibrationStatus = calibration ? "supported" : "unsupported";
346
- const opeRequested = options.requireOpe === true || options.ope !== void 0;
347
- const opeStatus = ope ? ope.support.supported ? "supported" : "unsupported" : opeRequested ? "unsupported" : "not_requested";
348
- if (!calibration) diagnostics.push("calibration unsupported: not enough confidence/outcome pairs");
349
- if (opeRequested && !opeTargetPolicy) diagnostics.push("OPE unsupported: missing target policy");
350
- else if (ope && !ope.support.supported)
351
- diagnostics.push(...ope.support.reasons.map((reason) => `OPE unsupported: ${reason}`));
352
- const status = overallStatus({
353
- selectiveStatus,
354
- hasCalibration: calibration !== null,
355
- opeStatus,
356
- opeRequested
357
- });
358
- return {
359
- policyId: options.policy.id,
360
- n: options.points.length,
361
- status,
362
- selectiveStatus,
363
- calibrationStatus,
364
- opeStatus,
365
- ...ope ? { opeTargetPolicyId: ope.targetPolicyId } : {},
366
- selective,
367
- ...calibration ? { calibration } : {},
368
- ...ope ? { ope } : {},
369
- diagnostics
370
- };
277
+ const selective = evaluateBeliefSelectivePolicy(options.points, options.policy, options.selective);
278
+ const calibration = calibrateBeliefDecisions(options.points, options.calibration);
279
+ const opeTargetPolicy = options.ope?.targetPolicy;
280
+ const ope = opeTargetPolicy ? evaluateBeliefOffPolicy(options.points, opeTargetPolicy, options.ope) : null;
281
+ const diagnostics = [];
282
+ const selectiveStatus = selective.recommendation;
283
+ const calibrationStatus = calibration ? "supported" : "unsupported";
284
+ const opeRequested = options.requireOpe === true || options.ope !== void 0;
285
+ const opeStatus = ope ? ope.support.supported ? "supported" : "unsupported" : opeRequested ? "unsupported" : "not_requested";
286
+ if (!calibration) diagnostics.push("calibration unsupported: not enough confidence/outcome pairs");
287
+ if (opeRequested && !opeTargetPolicy) diagnostics.push("OPE unsupported: missing target policy");
288
+ else if (ope && !ope.support.supported) diagnostics.push(...ope.support.reasons.map((reason) => `OPE unsupported: ${reason}`));
289
+ const status = overallStatus({
290
+ selectiveStatus,
291
+ hasCalibration: calibration !== null,
292
+ opeStatus,
293
+ opeRequested
294
+ });
295
+ return {
296
+ policyId: options.policy.id,
297
+ n: options.points.length,
298
+ status,
299
+ selectiveStatus,
300
+ calibrationStatus,
301
+ opeStatus,
302
+ ...ope ? { opeTargetPolicyId: ope.targetPolicyId } : {},
303
+ selective,
304
+ ...calibration ? { calibration } : {},
305
+ ...ope ? { ope } : {},
306
+ diagnostics
307
+ };
371
308
  }
372
309
  function overallStatus(options) {
373
- if (options.selectiveStatus === "need_more_data" || !options.hasCalibration) {
374
- return "need_more_data";
375
- }
376
- if (options.selectiveStatus === "hold") return "hold";
377
- if (options.opeRequested && options.opeStatus !== "supported") return "hold";
378
- return "ship";
379
- }
380
-
381
- // src/belief-state/code-agent-corpus.ts
382
- var FAILURE_RECOVERY_ACTIONS = ["retry", "verify", "continue", "stop"];
383
- var TARGET_LABELS = {
384
- "failure-recovery": "Failure recovery after tool or patch failure",
385
- "tool-selection": "Tool/action selection",
386
- "graph-completion": "Graph completion decision"
310
+ if (options.selectiveStatus === "need_more_data" || !options.hasCalibration) return "need_more_data";
311
+ if (options.selectiveStatus === "hold") return "hold";
312
+ if (options.opeRequested && options.opeStatus !== "supported") return "hold";
313
+ return "ship";
314
+ }
315
+ //#endregion
316
+ //#region src/belief-state/code-agent-corpus.ts
317
+ const FAILURE_RECOVERY_ACTIONS = [
318
+ "retry",
319
+ "verify",
320
+ "continue",
321
+ "stop"
322
+ ];
323
+ const TARGET_LABELS = {
324
+ "failure-recovery": "Failure recovery after tool or patch failure",
325
+ "tool-selection": "Tool/action selection",
326
+ "graph-completion": "Graph completion decision"
387
327
  };
388
328
  function extractCodeAgentBeliefDecisionPoints(options) {
389
- const entries = options.entries.filter(isRecord);
390
- const diagnostics = [];
391
- const observed = observedActionsFor(options.source, entries, options);
392
- const decisions = [];
393
- for (const action of observed) {
394
- if (action.kind === "tool" || action.kind === "patch") {
395
- decisions.push(toolSelectionDecision(action, options));
396
- }
397
- if (action.kind === "graph-completion") {
398
- decisions.push(graphCompletionDecision(action, options));
399
- }
400
- }
401
- for (const failed of observed) {
402
- if (failed.kind !== "tool" && failed.kind !== "patch" || failed.success !== false) continue;
403
- const next = observed.find(
404
- (candidate) => candidate.stepIndex > failed.stepIndex && (candidate.kind === "tool" || candidate.kind === "patch" || candidate.kind === "terminal")
405
- );
406
- if (!next) {
407
- diagnostics.push({
408
- runId: options.run.runId,
409
- severity: "warning",
410
- reason: `${failed.id}: failed action has no observable follow-up decision`
411
- });
412
- continue;
413
- }
414
- decisions.push(failureRecoveryDecision(failed, next, options));
415
- }
416
- if (decisions.length === 0) {
417
- diagnostics.push({
418
- runId: options.run.runId,
419
- severity: "info",
420
- reason: `no belief decision points extracted from ${options.source} entries`
421
- });
422
- }
423
- return { decisions, diagnostics };
329
+ const entries = options.entries.filter(isRecord$2);
330
+ const diagnostics = [];
331
+ const observed = observedActionsFor(options.source, entries, options);
332
+ const decisions = [];
333
+ for (const action of observed) {
334
+ if (action.kind === "tool" || action.kind === "patch") decisions.push(toolSelectionDecision(action, options));
335
+ if (action.kind === "graph-completion") decisions.push(graphCompletionDecision(action, options));
336
+ }
337
+ for (const failed of observed) {
338
+ if (failed.kind !== "tool" && failed.kind !== "patch" || failed.success !== false) continue;
339
+ const next = observed.find((candidate) => candidate.stepIndex > failed.stepIndex && (candidate.kind === "tool" || candidate.kind === "patch" || candidate.kind === "terminal"));
340
+ if (!next) {
341
+ diagnostics.push({
342
+ runId: options.run.runId,
343
+ severity: "warning",
344
+ reason: `${failed.id}: failed action has no observable follow-up decision`
345
+ });
346
+ continue;
347
+ }
348
+ decisions.push(failureRecoveryDecision(failed, next, options));
349
+ }
350
+ if (decisions.length === 0) diagnostics.push({
351
+ runId: options.run.runId,
352
+ severity: "info",
353
+ reason: `no belief decision points extracted from ${options.source} entries`
354
+ });
355
+ return {
356
+ decisions,
357
+ diagnostics
358
+ };
424
359
  }
425
360
  function inventoryBeliefDecisionPoints(points) {
426
- const byKind = [...groupBy(points, (point) => point.kind).entries()].map(
427
- ([kind, bucketPoints]) => bucketFor(kind, bucketPoints, { kind })
428
- ).sort(sortBuckets);
429
- const byTarget = [...groupBy(points, targetIdOf).entries()].filter((entry) => {
430
- return entry[0] !== void 0;
431
- }).map(([targetId, bucketPoints]) => bucketFor(targetId, bucketPoints, { targetId })).sort(sortBuckets);
432
- const diagnostics = [];
433
- if (points.length === 0) diagnostics.push("no decision points available");
434
- for (const bucket of byTarget) {
435
- if (bucket.withOutcome < bucket.n) {
436
- diagnostics.push(`${bucket.id}: ${bucket.n - bucket.withOutcome} decision(s) missing outcome`);
437
- }
438
- if (bucket.withBehaviorProb < bucket.n || bucket.withTargetProb < bucket.n) {
439
- diagnostics.push(`${bucket.id}: OPE support incomplete`);
440
- }
441
- }
442
- return { n: points.length, byKind, byTarget, diagnostics };
361
+ const byKind = [...groupBy(points, (point) => point.kind).entries()].map(([kind, bucketPoints]) => bucketFor(kind, bucketPoints, { kind })).sort(sortBuckets);
362
+ const byTarget = [...groupBy(points, targetIdOf).entries()].filter((entry) => {
363
+ return entry[0] !== void 0;
364
+ }).map(([targetId, bucketPoints]) => bucketFor(targetId, bucketPoints, { targetId })).sort(sortBuckets);
365
+ const diagnostics = [];
366
+ if (points.length === 0) diagnostics.push("no decision points available");
367
+ for (const bucket of byTarget) {
368
+ if (bucket.withOutcome < bucket.n) diagnostics.push(`${bucket.id}: ${bucket.n - bucket.withOutcome} decision(s) missing outcome`);
369
+ if (bucket.withBehaviorProb < bucket.n || bucket.withTargetProb < bucket.n) diagnostics.push(`${bucket.id}: OPE support incomplete`);
370
+ }
371
+ return {
372
+ n: points.length,
373
+ byKind,
374
+ byTarget,
375
+ diagnostics
376
+ };
443
377
  }
444
378
  function selectBeliefDecisionTarget(points, options = {}) {
445
- const minN = options.minN ?? 10;
446
- const minOutcomeCoverage = options.minOutcomeCoverage ?? 0.8;
447
- const preferredTargets = options.preferredTargets ?? [
448
- "failure-recovery",
449
- "tool-selection",
450
- "graph-completion"
451
- ];
452
- const inventory = inventoryBeliefDecisionPoints(points);
453
- for (const targetId of preferredTargets) {
454
- const support = inventory.byTarget.find((bucket) => bucket.targetId === targetId);
455
- if (!support) continue;
456
- const reasons = [];
457
- if (support.n < minN) reasons.push(`need at least ${minN} decisions, got ${support.n}`);
458
- const outcomeCoverage = support.n > 0 ? support.withOutcome / support.n : 0;
459
- if (outcomeCoverage < minOutcomeCoverage) {
460
- reasons.push(
461
- `outcome coverage ${outcomeCoverage.toFixed(2)} below ${minOutcomeCoverage.toFixed(2)}`
462
- );
463
- }
464
- if (reasons.length > 0) continue;
465
- const targetPoints = points.filter((point) => targetIdOf(point) === targetId);
466
- return {
467
- id: targetId,
468
- label: TARGET_LABELS[targetId],
469
- points: targetPoints,
470
- support,
471
- reasons
472
- };
473
- }
474
- return null;
379
+ const minN = options.minN ?? 10;
380
+ const minOutcomeCoverage = options.minOutcomeCoverage ?? .8;
381
+ const preferredTargets = options.preferredTargets ?? [
382
+ "failure-recovery",
383
+ "tool-selection",
384
+ "graph-completion"
385
+ ];
386
+ const inventory = inventoryBeliefDecisionPoints(points);
387
+ for (const targetId of preferredTargets) {
388
+ const support = inventory.byTarget.find((bucket) => bucket.targetId === targetId);
389
+ if (!support) continue;
390
+ const reasons = [];
391
+ if (support.n < minN) reasons.push(`need at least ${minN} decisions, got ${support.n}`);
392
+ const outcomeCoverage = support.n > 0 ? support.withOutcome / support.n : 0;
393
+ if (outcomeCoverage < minOutcomeCoverage) reasons.push(`outcome coverage ${outcomeCoverage.toFixed(2)} below ${minOutcomeCoverage.toFixed(2)}`);
394
+ if (reasons.length > 0) continue;
395
+ const targetPoints = points.filter((point) => targetIdOf(point) === targetId);
396
+ return {
397
+ id: targetId,
398
+ label: TARGET_LABELS[targetId],
399
+ points: targetPoints,
400
+ support,
401
+ reasons
402
+ };
403
+ }
404
+ return null;
475
405
  }
476
406
  function analyzeBeliefDecisionCorpus(options) {
477
- const inventory = inventoryBeliefDecisionPoints(options.points);
478
- const diagnostics = [...inventory.diagnostics];
479
- const target = options.targetId !== void 0 ? targetSelectionFor(options.points, options.targetId, options) : selectBeliefDecisionTarget(options.points, options);
480
- if (!target) {
481
- diagnostics.push("no decision target has enough support for policy evaluation");
482
- return { inventory, diagnostics };
483
- }
484
- const policy = options.policy ?? thresholdSelectivePolicy({
485
- id: `${target.id}:confidence>=${options.confidenceThreshold ?? 0.5}`,
486
- confidenceThreshold: options.confidenceThreshold ?? 0.5,
487
- belowThresholdAction: "verify"
488
- });
489
- const minN = options.minN ?? 10;
490
- const evaluation = analyzeBeliefPolicy({
491
- points: target.points,
492
- policy,
493
- selective: {
494
- minN,
495
- minAccepted: options.minAccepted ?? Math.min(5, minN),
496
- minUtilityDelta: 0,
497
- ...options.policyOptions?.selective ?? {}
498
- },
499
- calibration: {
500
- minPairs: Math.min(10, minN),
501
- policy,
502
- region: "all",
503
- ...options.policyOptions?.calibration ?? {}
504
- },
505
- ope: {
506
- targetPolicy: embeddedBeliefOpeTargetPolicy(`${target.id}:embedded-target-prob`),
507
- minEffectiveSampleSize: minN,
508
- ...options.policyOptions?.ope ?? {}
509
- },
510
- requireOpe: options.requireOpe ?? true
511
- });
512
- return { inventory, target, policy, evaluation, diagnostics };
407
+ const inventory = inventoryBeliefDecisionPoints(options.points);
408
+ const diagnostics = [...inventory.diagnostics];
409
+ const target = options.targetId !== void 0 ? targetSelectionFor(options.points, options.targetId, options) : selectBeliefDecisionTarget(options.points, options);
410
+ if (!target) {
411
+ diagnostics.push("no decision target has enough support for policy evaluation");
412
+ return {
413
+ inventory,
414
+ diagnostics
415
+ };
416
+ }
417
+ const policy = options.policy ?? thresholdSelectivePolicy({
418
+ id: `${target.id}:confidence>=${options.confidenceThreshold ?? .5}`,
419
+ confidenceThreshold: options.confidenceThreshold ?? .5,
420
+ belowThresholdAction: "verify"
421
+ });
422
+ const minN = options.minN ?? 10;
423
+ return {
424
+ inventory,
425
+ target,
426
+ policy,
427
+ evaluation: analyzeBeliefPolicy({
428
+ points: target.points,
429
+ policy,
430
+ selective: {
431
+ minN,
432
+ minAccepted: options.minAccepted ?? Math.min(5, minN),
433
+ minUtilityDelta: 0,
434
+ ...options.policyOptions?.selective ?? {}
435
+ },
436
+ calibration: {
437
+ minPairs: Math.min(10, minN),
438
+ policy,
439
+ region: "all",
440
+ ...options.policyOptions?.calibration ?? {}
441
+ },
442
+ ope: {
443
+ targetPolicy: embeddedBeliefOpeTargetPolicy(`${target.id}:embedded-target-prob`),
444
+ minEffectiveSampleSize: minN,
445
+ ...options.policyOptions?.ope ?? {}
446
+ },
447
+ requireOpe: options.requireOpe ?? true
448
+ }),
449
+ diagnostics
450
+ };
513
451
  }
514
452
  function observedActionsFor(source, entries, options) {
515
- const observation = options.observation ?? observeCodeAgentSession({ source, entries, sourcePath: options.sourcePath });
516
- if (observation.source !== source) {
517
- throw new Error("code-agent observation source does not match extraction source");
518
- }
519
- return observation.actions.map((action) => observedActionFromSession(action, options));
453
+ const observation = options.observation ?? observeCodeAgentSession({
454
+ source,
455
+ entries,
456
+ sourcePath: options.sourcePath
457
+ });
458
+ if (observation.source !== source) throw new Error("code-agent observation source does not match extraction source");
459
+ return observation.actions.map((action) => observedActionFromSession(action, options));
520
460
  }
521
461
  function observedActionFromSession(action, options) {
522
- return observedAction({
523
- options,
524
- localId: action.id,
525
- stepIndex: action.stepIndex,
526
- kind: action.kind,
527
- action: action.name,
528
- timestamp: action.timestampMs,
529
- success: action.status === "completed" ? true : action.status === "failed" ? false : void 0,
530
- costUsd: action.costUsd,
531
- metadata: { surface: action.surface, status: action.status, ...action.metadata }
532
- });
462
+ return observedAction({
463
+ options,
464
+ localId: action.id,
465
+ stepIndex: action.stepIndex,
466
+ kind: action.kind,
467
+ action: action.name,
468
+ timestamp: action.timestampMs,
469
+ success: action.status === "completed" ? true : action.status === "failed" ? false : void 0,
470
+ costUsd: action.costUsd,
471
+ metadata: {
472
+ surface: action.surface,
473
+ status: action.status,
474
+ ...action.metadata
475
+ }
476
+ });
533
477
  }
534
478
  function toolSelectionDecision(action, options) {
535
- return {
536
- id: `${options.run.runId}:tool-selection:${action.localId}`,
537
- runId: options.run.runId,
538
- scenarioId: options.run.scenarioId,
539
- stepIndex: action.stepIndex,
540
- kind: "tool-select",
541
- chosenAction: action.action,
542
- candidateActions: [action.action],
543
- confidence: 0.65,
544
- costUsd: action.costUsd,
545
- evidence: action.evidence,
546
- outcome: outcomeFromAction(action, options.run),
547
- metadata: {
548
- target: "tool-selection",
549
- source: options.source,
550
- actionKind: action.kind,
551
- confidenceSource: "fixed-observed-action-prior",
552
- ...action.metadata
553
- }
554
- };
479
+ return {
480
+ id: `${options.run.runId}:tool-selection:${action.localId}`,
481
+ runId: options.run.runId,
482
+ scenarioId: options.run.scenarioId,
483
+ stepIndex: action.stepIndex,
484
+ kind: "tool-select",
485
+ chosenAction: action.action,
486
+ candidateActions: [action.action],
487
+ confidence: .65,
488
+ costUsd: action.costUsd,
489
+ evidence: action.evidence,
490
+ outcome: outcomeFromAction(action, options.run),
491
+ metadata: {
492
+ target: "tool-selection",
493
+ source: options.source,
494
+ actionKind: action.kind,
495
+ confidenceSource: "fixed-observed-action-prior",
496
+ ...action.metadata
497
+ }
498
+ };
555
499
  }
556
500
  function graphCompletionDecision(action, options) {
557
- return {
558
- id: `${options.run.runId}:graph-completion:${action.localId}`,
559
- runId: options.run.runId,
560
- scenarioId: options.run.scenarioId,
561
- stepIndex: action.stepIndex,
562
- kind: "stop",
563
- chosenAction: "complete",
564
- candidateActions: ["complete", "continue", "verify"],
565
- confidence: 0.75,
566
- evidence: action.evidence,
567
- outcome: outcomeFromAction(action, options.run),
568
- metadata: {
569
- target: "graph-completion",
570
- source: options.source,
571
- confidenceSource: "fixed-graph-completion-prior",
572
- ...action.metadata
573
- }
574
- };
501
+ return {
502
+ id: `${options.run.runId}:graph-completion:${action.localId}`,
503
+ runId: options.run.runId,
504
+ scenarioId: options.run.scenarioId,
505
+ stepIndex: action.stepIndex,
506
+ kind: "stop",
507
+ chosenAction: "complete",
508
+ candidateActions: [
509
+ "complete",
510
+ "continue",
511
+ "verify"
512
+ ],
513
+ confidence: .75,
514
+ evidence: action.evidence,
515
+ outcome: outcomeFromAction(action, options.run),
516
+ metadata: {
517
+ target: "graph-completion",
518
+ source: options.source,
519
+ confidenceSource: "fixed-graph-completion-prior",
520
+ ...action.metadata
521
+ }
522
+ };
575
523
  }
576
524
  function failureRecoveryDecision(failed, next, options) {
577
- const chosenAction = classifyFailureRecovery(failed, next);
578
- return {
579
- id: `${options.run.runId}:failure-recovery:${failed.localId}`,
580
- runId: options.run.runId,
581
- scenarioId: options.run.scenarioId,
582
- stepIndex: failed.stepIndex,
583
- kind: "retry",
584
- chosenAction,
585
- candidateActions: [...FAILURE_RECOVERY_ACTIONS],
586
- confidence: recoveryConfidence(chosenAction),
587
- evidence: [...failed.evidence, ...next.evidence],
588
- outcome: outcomeFromAction(next, options.run),
589
- metadata: {
590
- target: "failure-recovery",
591
- source: options.source,
592
- failedActionKind: failed.kind,
593
- failedAction: failed.action,
594
- nextActionKind: next.kind,
595
- nextAction: next.action,
596
- confidenceSource: "heuristic-observed-follow-up"
597
- }
598
- };
525
+ const chosenAction = classifyFailureRecovery(failed, next);
526
+ return {
527
+ id: `${options.run.runId}:failure-recovery:${failed.localId}`,
528
+ runId: options.run.runId,
529
+ scenarioId: options.run.scenarioId,
530
+ stepIndex: failed.stepIndex,
531
+ kind: "retry",
532
+ chosenAction,
533
+ candidateActions: [...FAILURE_RECOVERY_ACTIONS],
534
+ confidence: recoveryConfidence(chosenAction),
535
+ evidence: [...failed.evidence, ...next.evidence],
536
+ outcome: outcomeFromAction(next, options.run),
537
+ metadata: {
538
+ target: "failure-recovery",
539
+ source: options.source,
540
+ failedActionKind: failed.kind,
541
+ failedAction: failed.action,
542
+ nextActionKind: next.kind,
543
+ nextAction: next.action,
544
+ confidenceSource: "heuristic-observed-follow-up"
545
+ }
546
+ };
599
547
  }
600
548
  function classifyFailureRecovery(failed, next) {
601
- if (next.kind === "terminal") return "stop";
602
- if (isVerificationAction(next.action)) return "verify";
603
- if (next.kind === failed.kind && next.action === failed.action) return "retry";
604
- return "continue";
549
+ if (next.kind === "terminal") return "stop";
550
+ if (isVerificationAction(next.action)) return "verify";
551
+ if (next.kind === failed.kind && next.action === failed.action) return "retry";
552
+ return "continue";
605
553
  }
606
554
  function recoveryConfidence(action) {
607
- if (action === "verify") return 0.8;
608
- if (action === "retry") return 0.6;
609
- if (action === "stop") return 0.55;
610
- return 0.35;
555
+ if (action === "verify") return .8;
556
+ if (action === "retry") return .6;
557
+ if (action === "stop") return .55;
558
+ return .35;
611
559
  }
612
560
  function isVerificationAction(action) {
613
- const normalized = action.toLowerCase();
614
- return normalized.includes("verify") || normalized.includes("test") || normalized.includes("check") || normalized.includes("lint") || normalized.includes("build") || normalized.includes("typecheck") || normalized.includes("pytest") || normalized.includes("vitest") || normalized.includes("tsc");
561
+ const normalized = action.toLowerCase();
562
+ return normalized.includes("verify") || normalized.includes("test") || normalized.includes("check") || normalized.includes("lint") || normalized.includes("build") || normalized.includes("typecheck") || normalized.includes("pytest") || normalized.includes("vitest") || normalized.includes("tsc");
615
563
  }
616
564
  function outcomeFromAction(action, run) {
617
- const runScore = scoreFromRun(run);
618
- const success = action.success ?? (runScore !== null ? runScore >= 0.5 : void 0);
619
- const score = action.success === void 0 ? runScore ?? void 0 : action.success ? 1 : 0;
620
- if (success === void 0 && score === void 0) return void 0;
621
- return {
622
- ...success !== void 0 ? { success } : {},
623
- ...score !== void 0 ? { score, reward: score } : {},
624
- ...action.costUsd !== void 0 ? { costUsd: action.costUsd } : {},
625
- metadata: {
626
- outcomeSource: action.success === void 0 ? "run-score" : "observed-action-status"
627
- }
628
- };
565
+ const runScore = scoreFromRun(run);
566
+ const success = action.success ?? (runScore !== null ? runScore >= .5 : void 0);
567
+ const score = action.success === void 0 ? runScore ?? void 0 : action.success ? 1 : 0;
568
+ if (success === void 0 && score === void 0) return void 0;
569
+ return {
570
+ ...success !== void 0 ? { success } : {},
571
+ ...score !== void 0 ? {
572
+ score,
573
+ reward: score
574
+ } : {},
575
+ ...action.costUsd !== void 0 ? { costUsd: action.costUsd } : {},
576
+ metadata: { outcomeSource: action.success === void 0 ? "run-score" : "observed-action-status" }
577
+ };
629
578
  }
630
579
  function observedAction(input) {
631
- const id = `${input.options.run.runId}:${input.options.source}:${input.localId}`;
632
- return {
633
- id,
634
- localId: input.localId,
635
- stepIndex: input.stepIndex,
636
- kind: input.kind,
637
- action: input.action,
638
- timestamp: input.timestamp,
639
- success: input.success,
640
- costUsd: input.costUsd,
641
- evidence: [
642
- {
643
- source: "event",
644
- id,
645
- runId: input.options.run.runId,
646
- detail: input.action,
647
- metadata: {
648
- source: input.options.source,
649
- sourcePath: input.options.sourcePath,
650
- ...input.metadata
651
- }
652
- }
653
- ],
654
- metadata: input.metadata ?? {}
655
- };
580
+ const id = `${input.options.run.runId}:${input.options.source}:${input.localId}`;
581
+ return {
582
+ id,
583
+ localId: input.localId,
584
+ stepIndex: input.stepIndex,
585
+ kind: input.kind,
586
+ action: input.action,
587
+ timestamp: input.timestamp,
588
+ success: input.success,
589
+ costUsd: input.costUsd,
590
+ evidence: [{
591
+ source: "event",
592
+ id,
593
+ runId: input.options.run.runId,
594
+ detail: input.action,
595
+ metadata: {
596
+ source: input.options.source,
597
+ sourcePath: input.options.sourcePath,
598
+ ...input.metadata
599
+ }
600
+ }],
601
+ metadata: input.metadata ?? {}
602
+ };
656
603
  }
657
604
  function targetSelectionFor(points, targetId, options) {
658
- const targetPoints = points.filter((point) => targetIdOf(point) === targetId);
659
- if (targetPoints.length === 0) return null;
660
- const support = bucketFor(targetId, targetPoints, { targetId });
661
- const minN = options.minN ?? 10;
662
- const minOutcomeCoverage = options.minOutcomeCoverage ?? 0.8;
663
- const reasons = [];
664
- if (support.n < minN) reasons.push(`need at least ${minN} decisions, got ${support.n}`);
665
- const outcomeCoverage = support.n > 0 ? support.withOutcome / support.n : 0;
666
- if (outcomeCoverage < minOutcomeCoverage) {
667
- reasons.push(
668
- `outcome coverage ${outcomeCoverage.toFixed(2)} below ${minOutcomeCoverage.toFixed(2)}`
669
- );
670
- }
671
- if (reasons.length > 0) return null;
672
- return { id: targetId, label: TARGET_LABELS[targetId], points: targetPoints, support, reasons };
605
+ const targetPoints = points.filter((point) => targetIdOf(point) === targetId);
606
+ if (targetPoints.length === 0) return null;
607
+ const support = bucketFor(targetId, targetPoints, { targetId });
608
+ const minN = options.minN ?? 10;
609
+ const minOutcomeCoverage = options.minOutcomeCoverage ?? .8;
610
+ const reasons = [];
611
+ if (support.n < minN) reasons.push(`need at least ${minN} decisions, got ${support.n}`);
612
+ const outcomeCoverage = support.n > 0 ? support.withOutcome / support.n : 0;
613
+ if (outcomeCoverage < minOutcomeCoverage) reasons.push(`outcome coverage ${outcomeCoverage.toFixed(2)} below ${minOutcomeCoverage.toFixed(2)}`);
614
+ if (reasons.length > 0) return null;
615
+ return {
616
+ id: targetId,
617
+ label: TARGET_LABELS[targetId],
618
+ points: targetPoints,
619
+ support,
620
+ reasons
621
+ };
673
622
  }
674
623
  function bucketFor(id, points, identity) {
675
- const outcomes = points.filter((point) => point.outcome);
676
- const scores = outcomes.map((point) => outcomeScore2(point.outcome)).filter((score) => score !== null);
677
- const confidences = points.map((point) => point.confidence).filter((confidence) => typeof confidence === "number");
678
- const successes = outcomes.filter((point) => point.outcome?.success === true).length;
679
- const successDenominator = outcomes.filter(
680
- (point) => typeof point.outcome?.success === "boolean"
681
- ).length;
682
- return {
683
- id,
684
- ...identity,
685
- n: points.length,
686
- withOutcome: outcomes.length,
687
- withConfidence: confidences.length,
688
- withCandidateActions: points.filter((point) => (point.candidateActions?.length ?? 0) > 0).length,
689
- withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
690
- withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
691
- successRate: successDenominator > 0 ? successes / successDenominator : null,
692
- meanScore: scores.length > 0 ? mean2(scores) : null,
693
- meanConfidence: confidences.length > 0 ? mean2(confidences) : null
694
- };
624
+ const outcomes = points.filter((point) => point.outcome);
625
+ const scores = outcomes.map((point) => outcomeScore(point.outcome)).filter((score) => score !== null);
626
+ const confidences = points.map((point) => point.confidence).filter((confidence) => typeof confidence === "number");
627
+ const successes = outcomes.filter((point) => point.outcome?.success === true).length;
628
+ const successDenominator = outcomes.filter((point) => typeof point.outcome?.success === "boolean").length;
629
+ return {
630
+ id,
631
+ ...identity,
632
+ n: points.length,
633
+ withOutcome: outcomes.length,
634
+ withConfidence: confidences.length,
635
+ withCandidateActions: points.filter((point) => (point.candidateActions?.length ?? 0) > 0).length,
636
+ withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
637
+ withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
638
+ successRate: successDenominator > 0 ? successes / successDenominator : null,
639
+ meanScore: scores.length > 0 ? mean$1(scores) : null,
640
+ meanConfidence: confidences.length > 0 ? mean$1(confidences) : null
641
+ };
695
642
  }
696
643
  function targetIdOf(point) {
697
- const target = point.metadata?.target;
698
- if (target === "failure-recovery" || target === "tool-selection" || target === "graph-completion")
699
- return target;
700
- return void 0;
701
- }
702
- function outcomeScore2(outcome) {
703
- if (!outcome) return null;
704
- if (typeof outcome.score === "number") return outcome.score;
705
- if (typeof outcome.reward === "number") return outcome.reward;
706
- if (outcome.success === true) return 1;
707
- if (outcome.success === false) return 0;
708
- return null;
709
- }
644
+ const target = point.metadata?.target;
645
+ if (target === "failure-recovery" || target === "tool-selection" || target === "graph-completion") return target;
646
+ }
647
+ function outcomeScore(outcome) {
648
+ if (!outcome) return null;
649
+ if (typeof outcome.score === "number") return outcome.score;
650
+ if (typeof outcome.reward === "number") return outcome.reward;
651
+ if (outcome.success === true) return 1;
652
+ if (outcome.success === false) return 0;
653
+ return null;
654
+ }
655
+ /**
656
+ * GATED (`trainingScore`). The number this returns becomes a belief-decision
657
+ * point's `outcome.score` AND its `outcome.reward` — corpus labels, i.e.
658
+ * training data by another name. A run flagged as gamed would otherwise label
659
+ * every decision on its trajectory a success and teach a belief model to
660
+ * predict that the gaming path works.
661
+ */
710
662
  function scoreFromRun(run) {
711
- if (typeof run.outcome.holdoutScore === "number") return run.outcome.holdoutScore;
712
- if (typeof run.outcome.searchScore === "number") return run.outcome.searchScore;
713
- return null;
663
+ return trainingScore(run) ?? null;
714
664
  }
715
665
  function sortBuckets(a, b) {
716
- return b.n - a.n || a.id.localeCompare(b.id);
666
+ return b.n - a.n || a.id.localeCompare(b.id);
717
667
  }
718
668
  function groupBy(values, keyOf) {
719
- const map = /* @__PURE__ */ new Map();
720
- for (const value of values) {
721
- const key = keyOf(value);
722
- const bucket = map.get(key);
723
- if (bucket) bucket.push(value);
724
- else map.set(key, [value]);
725
- }
726
- return map;
727
- }
728
- function mean2(values) {
729
- return values.reduce((sum, value) => sum + value, 0) / values.length;
730
- }
731
- function isRecord(value) {
732
- return value !== null && typeof value === "object" && !Array.isArray(value);
733
- }
734
-
735
- // src/belief-state/research-evidence.ts
669
+ const map = /* @__PURE__ */ new Map();
670
+ for (const value of values) {
671
+ const key = keyOf(value);
672
+ const bucket = map.get(key);
673
+ if (bucket) bucket.push(value);
674
+ else map.set(key, [value]);
675
+ }
676
+ return map;
677
+ }
678
+ function mean$1(values) {
679
+ return values.reduce((sum, value) => sum + value, 0) / values.length;
680
+ }
681
+ function isRecord$2(value) {
682
+ return value !== null && typeof value === "object" && !Array.isArray(value);
683
+ }
684
+ //#endregion
685
+ //#region src/belief-state/research-evidence.ts
736
686
  function buildBeliefDecisionResearchEvidencePacket(options) {
737
- const claimScope = options.claimScope ?? "counterfactual";
738
- const requireOpe = claimScope === "counterfactual";
739
- const analysis = analyzeBeliefDecisionCorpus({
740
- ...options,
741
- requireOpe: options.requireOpe ?? requireOpe
742
- });
743
- const gates = [
744
- corpusGate(analysis),
745
- selectiveGate(analysis),
746
- calibrationGate(analysis),
747
- ...requireOpe ? [opeGate(analysis)] : []
748
- ];
749
- const caveats = unique([
750
- ...gates.flatMap((gate) => gate.caveats),
751
- ...claimScope === "selective" ? ["counterfactual claims excluded: OPE support was not required"] : []
752
- ]);
753
- return {
754
- claimScope,
755
- status: gates.every((gate) => gate.status === "supported") ? "supported" : "blocked",
756
- analysis,
757
- gates,
758
- blockers: unique(gates.flatMap((gate) => gate.blockers)),
759
- caveats
760
- };
687
+ const claimScope = options.claimScope ?? "counterfactual";
688
+ const requireOpe = claimScope === "counterfactual";
689
+ const analysis = analyzeBeliefDecisionCorpus({
690
+ ...options,
691
+ requireOpe: options.requireOpe ?? requireOpe
692
+ });
693
+ const gates = [
694
+ corpusGate(analysis),
695
+ selectiveGate(analysis),
696
+ calibrationGate(analysis),
697
+ ...requireOpe ? [opeGate(analysis)] : []
698
+ ];
699
+ const caveats = unique([...gates.flatMap((gate) => gate.caveats), ...claimScope === "selective" ? ["counterfactual claims excluded: OPE support was not required"] : []]);
700
+ return {
701
+ claimScope,
702
+ status: gates.every((gate) => gate.status === "supported") ? "supported" : "blocked",
703
+ analysis,
704
+ gates,
705
+ blockers: unique(gates.flatMap((gate) => gate.blockers)),
706
+ caveats
707
+ };
761
708
  }
762
709
  function corpusGate(analysis) {
763
- const support = analysis.target?.support;
764
- if (!support) {
765
- return blocked("corpus", "no decision target has enough outcome support");
766
- }
767
- const caveats = support.withBehaviorProb < support.n || support.withTargetProb < support.n ? ["propensity support incomplete; counterfactual claims will require OPE support"] : [];
768
- return { id: "corpus", status: "supported", blockers: [], caveats };
710
+ const support = analysis.target?.support;
711
+ if (!support) return blocked("corpus", "no decision target has enough outcome support");
712
+ return {
713
+ id: "corpus",
714
+ status: "supported",
715
+ blockers: [],
716
+ caveats: support.withBehaviorProb < support.n || support.withTargetProb < support.n ? ["propensity support incomplete; counterfactual claims will require OPE support"] : []
717
+ };
769
718
  }
770
719
  function selectiveGate(analysis) {
771
- const evaluation = analysis.evaluation;
772
- if (!evaluation) return blocked("selective", "no policy evaluation was produced");
773
- if (evaluation.selectiveStatus !== "ship") {
774
- return blocked(
775
- "selective",
776
- ...orDefault(
777
- evaluation.selective.reasons,
778
- `selective status is ${evaluation.selectiveStatus}`
779
- )
780
- );
781
- }
782
- return { id: "selective", status: "supported", blockers: [], caveats: [] };
720
+ const evaluation = analysis.evaluation;
721
+ if (!evaluation) return blocked("selective", "no policy evaluation was produced");
722
+ if (evaluation.selectiveStatus !== "ship") return blocked("selective", ...orDefault(evaluation.selective.reasons, `selective status is ${evaluation.selectiveStatus}`));
723
+ return {
724
+ id: "selective",
725
+ status: "supported",
726
+ blockers: [],
727
+ caveats: []
728
+ };
783
729
  }
784
730
  function calibrationGate(analysis) {
785
- const evaluation = analysis.evaluation;
786
- if (!evaluation) return blocked("calibration", "no policy evaluation was produced");
787
- if (evaluation.calibrationStatus !== "supported") {
788
- return blocked("calibration", "not enough confidence/outcome pairs for calibration");
789
- }
790
- return { id: "calibration", status: "supported", blockers: [], caveats: [] };
731
+ const evaluation = analysis.evaluation;
732
+ if (!evaluation) return blocked("calibration", "no policy evaluation was produced");
733
+ if (evaluation.calibrationStatus !== "supported") return blocked("calibration", "not enough confidence/outcome pairs for calibration");
734
+ return {
735
+ id: "calibration",
736
+ status: "supported",
737
+ blockers: [],
738
+ caveats: []
739
+ };
791
740
  }
792
741
  function opeGate(analysis) {
793
- const evaluation = analysis.evaluation;
794
- if (!evaluation) return blocked("ope", "no policy evaluation was produced");
795
- if (evaluation.opeStatus !== "supported") {
796
- const reasons = evaluation.ope?.support.reasons ?? evaluation.diagnostics.filter((diagnostic) => diagnostic.includes("OPE"));
797
- return blocked("ope", ...orDefault(reasons, "missing OPE support"));
798
- }
799
- return { id: "ope", status: "supported", blockers: [], caveats: [] };
742
+ const evaluation = analysis.evaluation;
743
+ if (!evaluation) return blocked("ope", "no policy evaluation was produced");
744
+ if (evaluation.opeStatus !== "supported") return blocked("ope", ...orDefault(evaluation.ope?.support.reasons ?? evaluation.diagnostics.filter((diagnostic) => diagnostic.includes("OPE")), "missing OPE support"));
745
+ return {
746
+ id: "ope",
747
+ status: "supported",
748
+ blockers: [],
749
+ caveats: []
750
+ };
800
751
  }
801
752
  function blocked(id, ...blockers) {
802
- return { id, status: "blocked", blockers, caveats: [] };
753
+ return {
754
+ id,
755
+ status: "blocked",
756
+ blockers,
757
+ caveats: []
758
+ };
803
759
  }
804
760
  function orDefault(values, fallback) {
805
- return values.length > 0 ? values : [fallback];
761
+ return values.length > 0 ? values : [fallback];
806
762
  }
807
763
  function unique(values) {
808
- return [...new Set(values)];
764
+ return [...new Set(values)];
809
765
  }
810
-
811
- // src/belief-state/code-agent-evidence.ts
766
+ //#endregion
767
+ //#region src/belief-state/code-agent-evidence.ts
812
768
  function buildCodeAgentBeliefEvidenceCorpus(options) {
813
- const { sessions, ...evidenceOptions } = options;
814
- const runs = [];
815
- const metrics = [];
816
- const intakeDiagnostics = [];
817
- const extractionDiagnostics = [];
818
- const decisions = [];
819
- for (const session of sessions) {
820
- const intake = fromCodeAgentBeliefSession(session);
821
- runs.push(...intake.runs);
822
- metrics.push(...intake.metrics);
823
- intakeDiagnostics.push(...intake.diagnostics);
824
- for (const [index, run] of intake.runs.entries()) {
825
- const extraction = extractCodeAgentBeliefDecisionPoints({
826
- source: session.source,
827
- entries: session.entries,
828
- observation: intake.observations[index],
829
- run,
830
- sourcePath: session.sourcePath
831
- });
832
- decisions.push(...extraction.decisions);
833
- extractionDiagnostics.push(...extraction.diagnostics);
834
- }
835
- }
836
- const evidence = buildBeliefDecisionResearchEvidencePacket({
837
- ...evidenceOptions,
838
- points: decisions
839
- });
840
- return {
841
- runs,
842
- metrics,
843
- intakeDiagnostics,
844
- extractionDiagnostics,
845
- decisions,
846
- inventory: inventoryBeliefDecisionPoints(decisions),
847
- evidence
848
- };
769
+ const { sessions, ...evidenceOptions } = options;
770
+ const runs = [];
771
+ const metrics = [];
772
+ const intakeDiagnostics = [];
773
+ const extractionDiagnostics = [];
774
+ const decisions = [];
775
+ for (const session of sessions) {
776
+ const intake = fromCodeAgentBeliefSession(session);
777
+ runs.push(...intake.runs);
778
+ metrics.push(...intake.metrics);
779
+ intakeDiagnostics.push(...intake.diagnostics);
780
+ for (const [index, run] of intake.runs.entries()) {
781
+ const extraction = extractCodeAgentBeliefDecisionPoints({
782
+ source: session.source,
783
+ entries: session.entries,
784
+ observation: intake.observations[index],
785
+ run,
786
+ sourcePath: session.sourcePath
787
+ });
788
+ decisions.push(...extraction.decisions);
789
+ extractionDiagnostics.push(...extraction.diagnostics);
790
+ }
791
+ }
792
+ const evidence = buildBeliefDecisionResearchEvidencePacket({
793
+ ...evidenceOptions,
794
+ points: decisions
795
+ });
796
+ return {
797
+ runs,
798
+ metrics,
799
+ intakeDiagnostics,
800
+ extractionDiagnostics,
801
+ decisions,
802
+ inventory: inventoryBeliefDecisionPoints(decisions),
803
+ evidence
804
+ };
849
805
  }
850
806
  function fromCodeAgentBeliefSession(session) {
851
- switch (session.source) {
852
- case "codex":
853
- return fromCodexSession(session);
854
- case "claude-code":
855
- return fromClaudeCodeSession(session);
856
- case "opencode":
857
- return fromOpenCodeSession(session);
858
- case "kimi-code":
859
- return fromKimiCodeSession(session);
860
- case "pi":
861
- return fromPiSession(session);
862
- }
863
- }
864
-
865
- // src/belief-state/types.ts
866
- var BELIEF_DECISION_KINDS = [
867
- "continue",
868
- "verify",
869
- "ask",
870
- "retry",
871
- "stop",
872
- "memory-write",
873
- "memory-read",
874
- "tool-select",
875
- "skill-select",
876
- "workflow-select",
877
- "surface-promote"
807
+ switch (session.source) {
808
+ case "codex": return fromCodexSession(session);
809
+ case "claude-code": return fromClaudeCodeSession(session);
810
+ case "opencode": return fromOpenCodeSession(session);
811
+ case "kimi-code": return fromKimiCodeSession(session);
812
+ case "pi": return fromPiSession(session);
813
+ }
814
+ }
815
+ //#endregion
816
+ //#region src/belief-state/types.ts
817
+ const BELIEF_DECISION_KINDS = [
818
+ "continue",
819
+ "verify",
820
+ "ask",
821
+ "retry",
822
+ "stop",
823
+ "memory-write",
824
+ "memory-read",
825
+ "tool-select",
826
+ "skill-select",
827
+ "workflow-select",
828
+ "surface-promote"
878
829
  ];
879
- var BELIEF_EVIDENCE_SOURCES = [
880
- "run",
881
- "span",
882
- "event",
883
- "finding",
884
- "memory",
885
- "knowledge",
886
- "policy"
830
+ const BELIEF_EVIDENCE_SOURCES = [
831
+ "run",
832
+ "span",
833
+ "event",
834
+ "finding",
835
+ "memory",
836
+ "knowledge",
837
+ "policy"
887
838
  ];
888
- var BELIEF_EVIDENCE_QUALITIES = [
889
- "direct",
890
- "derived",
891
- "self-reported",
892
- "unverified",
893
- "stale",
894
- "contradicted"
839
+ const BELIEF_EVIDENCE_QUALITIES = [
840
+ "direct",
841
+ "derived",
842
+ "self-reported",
843
+ "unverified",
844
+ "stale",
845
+ "contradicted"
895
846
  ];
896
- var BELIEF_EVALUATION_CRITERIA = [
897
- {
898
- id: "capture-integrity",
899
- label: "Capture integrity",
900
- reasonCodes: ["trace-missing", "run-record-missing", "backend-integrity-missing"]
901
- },
902
- {
903
- id: "decision-completeness",
904
- label: "Decision completeness",
905
- reasonCodes: [
906
- "candidate-actions-missing",
907
- "chosen-action-missing",
908
- "decision-evidence-missing"
909
- ]
910
- },
911
- {
912
- id: "evidence-quality",
913
- label: "Evidence quality",
914
- reasonCodes: [
915
- "evidence-stale",
916
- "evidence-contradictory",
917
- "evidence-unverified",
918
- "evidence-self-reported"
919
- ]
920
- },
921
- {
922
- id: "outcome-quality",
923
- label: "Outcome quality",
924
- reasonCodes: ["outcome-missing", "outcome-delayed", "cost-missing"]
925
- },
926
- {
927
- id: "calibration",
928
- label: "Calibration",
929
- reasonCodes: ["confidence-missing", "calibration-unsupported", "calibration-gap-high"]
930
- },
931
- {
932
- id: "accepted-region-risk",
933
- label: "Accepted-region risk",
934
- reasonCodes: ["accepted-error-high", "coverage-too-low"]
935
- },
936
- {
937
- id: "policy-value",
938
- label: "Policy value",
939
- reasonCodes: ["utility-lift-missing", "baseline-dominates", "cost-too-high"]
940
- },
941
- {
942
- id: "ope-support",
943
- label: "OPE support",
944
- reasonCodes: [
945
- "behavior-propensity-missing",
946
- "behavior-propensity-invalid",
947
- "target-propensity-missing",
948
- "target-propensity-invalid",
949
- "effective-sample-size-low",
950
- "importance-weight-high"
951
- ]
952
- },
953
- {
954
- id: "memory-health",
955
- label: "Memory health",
956
- reasonCodes: [
957
- "memory-stale",
958
- "memory-poisoning-risk",
959
- "context-bloat",
960
- "memory-write-unverified"
961
- ]
962
- },
963
- {
964
- id: "surface-attribution",
965
- label: "Surface attribution",
966
- reasonCodes: ["surface-claim-unsupported", "causal-attribution-missing"]
967
- },
968
- {
969
- id: "generalization",
970
- label: "Generalization",
971
- reasonCodes: [
972
- "split-missing",
973
- "holdout-regression",
974
- "task-family-coverage-low",
975
- "leakage-risk"
976
- ]
977
- },
978
- {
979
- id: "promotion",
980
- label: "Promotion",
981
- reasonCodes: ["negative-control-failed", "promotion-gate-failed", "human-review-required"]
982
- }
847
+ const BELIEF_EVALUATION_CRITERIA = [
848
+ {
849
+ id: "capture-integrity",
850
+ label: "Capture integrity",
851
+ reasonCodes: [
852
+ "trace-missing",
853
+ "run-record-missing",
854
+ "backend-integrity-missing"
855
+ ]
856
+ },
857
+ {
858
+ id: "decision-completeness",
859
+ label: "Decision completeness",
860
+ reasonCodes: [
861
+ "candidate-actions-missing",
862
+ "chosen-action-missing",
863
+ "decision-evidence-missing"
864
+ ]
865
+ },
866
+ {
867
+ id: "evidence-quality",
868
+ label: "Evidence quality",
869
+ reasonCodes: [
870
+ "evidence-stale",
871
+ "evidence-contradictory",
872
+ "evidence-unverified",
873
+ "evidence-self-reported"
874
+ ]
875
+ },
876
+ {
877
+ id: "outcome-quality",
878
+ label: "Outcome quality",
879
+ reasonCodes: [
880
+ "outcome-missing",
881
+ "outcome-delayed",
882
+ "cost-missing"
883
+ ]
884
+ },
885
+ {
886
+ id: "calibration",
887
+ label: "Calibration",
888
+ reasonCodes: [
889
+ "confidence-missing",
890
+ "calibration-unsupported",
891
+ "calibration-gap-high"
892
+ ]
893
+ },
894
+ {
895
+ id: "accepted-region-risk",
896
+ label: "Accepted-region risk",
897
+ reasonCodes: ["accepted-error-high", "coverage-too-low"]
898
+ },
899
+ {
900
+ id: "policy-value",
901
+ label: "Policy value",
902
+ reasonCodes: [
903
+ "utility-lift-missing",
904
+ "baseline-dominates",
905
+ "cost-too-high"
906
+ ]
907
+ },
908
+ {
909
+ id: "ope-support",
910
+ label: "OPE support",
911
+ reasonCodes: [
912
+ "behavior-propensity-missing",
913
+ "behavior-propensity-invalid",
914
+ "target-propensity-missing",
915
+ "target-propensity-invalid",
916
+ "effective-sample-size-low",
917
+ "importance-weight-high"
918
+ ]
919
+ },
920
+ {
921
+ id: "memory-health",
922
+ label: "Memory health",
923
+ reasonCodes: [
924
+ "memory-stale",
925
+ "memory-poisoning-risk",
926
+ "context-bloat",
927
+ "memory-write-unverified"
928
+ ]
929
+ },
930
+ {
931
+ id: "surface-attribution",
932
+ label: "Surface attribution",
933
+ reasonCodes: ["surface-claim-unsupported", "causal-attribution-missing"]
934
+ },
935
+ {
936
+ id: "generalization",
937
+ label: "Generalization",
938
+ reasonCodes: [
939
+ "split-missing",
940
+ "holdout-regression",
941
+ "task-family-coverage-low",
942
+ "leakage-risk"
943
+ ]
944
+ },
945
+ {
946
+ id: "promotion",
947
+ label: "Promotion",
948
+ reasonCodes: [
949
+ "negative-control-failed",
950
+ "promotion-gate-failed",
951
+ "human-review-required"
952
+ ]
953
+ }
983
954
  ];
984
955
  function isBeliefDecisionKind(value) {
985
- return typeof value === "string" && BELIEF_DECISION_KINDS.includes(value);
956
+ return typeof value === "string" && BELIEF_DECISION_KINDS.includes(value);
986
957
  }
987
958
  function isBeliefEvidenceSource(value) {
988
- return typeof value === "string" && BELIEF_EVIDENCE_SOURCES.includes(value);
989
- }
990
-
991
- // src/belief-state/extract.ts
992
- var DECISION_MARKERS = /* @__PURE__ */ new Set(["belief_decision", "belief.decision", "decision_point"]);
959
+ return typeof value === "string" && BELIEF_EVIDENCE_SOURCES.includes(value);
960
+ }
961
+ //#endregion
962
+ //#region src/belief-state/extract.ts
963
+ const DECISION_MARKERS = /* @__PURE__ */ new Set([
964
+ "belief_decision",
965
+ "belief.decision",
966
+ "decision_point"
967
+ ]);
993
968
  async function extractBeliefDecisionPoints(store, options = {}) {
994
- const runs = options.runIds ? (await Promise.all(options.runIds.map((runId) => store.getRun(runId)))).filter(Boolean) : await store.listRuns();
995
- const decisions = [];
996
- const diagnostics = [];
997
- for (const run of runs) {
998
- if (!run) continue;
999
- const events = await store.events({ runId: run.runId });
1000
- const spans = await store.spans({ runId: run.runId });
1001
- const spanIds = new Set(spans.map((span) => span.spanId));
1002
- let stepIndex = 0;
1003
- for (const event of [...events].sort((a, b) => a.timestamp - b.timestamp)) {
1004
- const parsed = parseDecisionEvent(event, {
1005
- scenarioId: run.scenarioId,
1006
- stepIndex,
1007
- spanExists: event.spanId ? spanIds.has(event.spanId) : false
1008
- });
1009
- if (!parsed) continue;
1010
- if ("diagnostic" in parsed) {
1011
- diagnostics.push(parsed.diagnostic);
1012
- continue;
1013
- }
1014
- decisions.push(parsed.decision);
1015
- stepIndex++;
1016
- }
1017
- }
1018
- return { decisions, diagnostics };
969
+ const runs = options.runIds ? (await Promise.all(options.runIds.map((runId) => store.getRun(runId)))).filter(Boolean) : await store.listRuns();
970
+ const decisions = [];
971
+ const diagnostics = [];
972
+ for (const run of runs) {
973
+ if (!run) continue;
974
+ const events = await store.events({ runId: run.runId });
975
+ const spans = await store.spans({ runId: run.runId });
976
+ const spanIds = new Set(spans.map((span) => span.spanId));
977
+ let stepIndex = 0;
978
+ for (const event of [...events].sort((a, b) => a.timestamp - b.timestamp)) {
979
+ const parsed = parseDecisionEvent(event, {
980
+ scenarioId: run.scenarioId,
981
+ stepIndex,
982
+ spanExists: event.spanId ? spanIds.has(event.spanId) : false
983
+ });
984
+ if (!parsed) continue;
985
+ if ("diagnostic" in parsed) {
986
+ diagnostics.push(parsed.diagnostic);
987
+ continue;
988
+ }
989
+ decisions.push(parsed.decision);
990
+ stepIndex++;
991
+ }
992
+ }
993
+ return {
994
+ decisions,
995
+ diagnostics
996
+ };
1019
997
  }
1020
998
  function parseDecisionEvent(event, context) {
1021
- const payload = event.payload;
1022
- const marker = stringField(payload, "kind") ?? stringField(payload, "type");
1023
- if (!marker || !DECISION_MARKERS.has(marker)) return null;
1024
- const decisionKind = stringField(payload, "decisionKind");
1025
- if (!isBeliefDecisionKind(decisionKind)) {
1026
- return {
1027
- diagnostic: {
1028
- runId: event.runId,
1029
- eventId: event.eventId,
1030
- severity: "warning",
1031
- reason: `belief decision event has unsupported decisionKind "${decisionKind ?? ""}"`
1032
- }
1033
- };
1034
- }
1035
- const chosenAction = stringField(payload, "chosenAction");
1036
- if (!chosenAction) {
1037
- return {
1038
- diagnostic: {
1039
- runId: event.runId,
1040
- eventId: event.eventId,
1041
- severity: "warning",
1042
- reason: "belief decision event is missing chosenAction"
1043
- }
1044
- };
1045
- }
1046
- const evidence = [
1047
- {
1048
- source: "event",
1049
- id: event.eventId,
1050
- runId: event.runId,
1051
- eventId: event.eventId,
1052
- quality: "direct"
1053
- }
1054
- ];
1055
- if (event.spanId && context.spanExists) {
1056
- evidence.push({
1057
- source: "span",
1058
- id: event.spanId,
1059
- runId: event.runId,
1060
- spanId: event.spanId,
1061
- quality: "direct"
1062
- });
1063
- }
1064
- return {
1065
- decision: {
1066
- id: stringField(payload, "id") ?? event.eventId,
1067
- runId: event.runId,
1068
- scenarioId: stringField(payload, "scenarioId") ?? context.scenarioId,
1069
- stepIndex: numberField(payload, "stepIndex") ?? context.stepIndex,
1070
- kind: decisionKind,
1071
- chosenAction,
1072
- candidateActions: stringArrayField(payload, "candidateActions"),
1073
- confidence: finiteUnitField(payload, "confidence"),
1074
- behaviorProb: numberField(payload, "behaviorProb"),
1075
- targetProb: numberField(payload, "targetProb"),
1076
- qHatChosen: finiteUnitField(payload, "qHatChosen"),
1077
- vHatTarget: finiteUnitField(payload, "vHatTarget"),
1078
- qHat: finiteUnitField(payload, "qHat"),
1079
- costUsd: nonNegativeNumberField(payload, "costUsd"),
1080
- evidence,
1081
- outcome: parseOutcome(payload),
1082
- metadata: recordField(payload, "metadata")
1083
- }
1084
- };
999
+ const payload = event.payload;
1000
+ const marker = stringField(payload, "kind") ?? stringField(payload, "type");
1001
+ if (!marker || !DECISION_MARKERS.has(marker)) return null;
1002
+ const decisionKind = stringField(payload, "decisionKind");
1003
+ if (!isBeliefDecisionKind(decisionKind)) return { diagnostic: {
1004
+ runId: event.runId,
1005
+ eventId: event.eventId,
1006
+ severity: "warning",
1007
+ reason: `belief decision event has unsupported decisionKind "${decisionKind ?? ""}"`
1008
+ } };
1009
+ const chosenAction = stringField(payload, "chosenAction");
1010
+ if (!chosenAction) return { diagnostic: {
1011
+ runId: event.runId,
1012
+ eventId: event.eventId,
1013
+ severity: "warning",
1014
+ reason: "belief decision event is missing chosenAction"
1015
+ } };
1016
+ const evidence = [{
1017
+ source: "event",
1018
+ id: event.eventId,
1019
+ runId: event.runId,
1020
+ eventId: event.eventId,
1021
+ quality: "direct"
1022
+ }];
1023
+ if (event.spanId && context.spanExists) evidence.push({
1024
+ source: "span",
1025
+ id: event.spanId,
1026
+ runId: event.runId,
1027
+ spanId: event.spanId,
1028
+ quality: "direct"
1029
+ });
1030
+ return { decision: {
1031
+ id: stringField(payload, "id") ?? event.eventId,
1032
+ runId: event.runId,
1033
+ scenarioId: stringField(payload, "scenarioId") ?? context.scenarioId,
1034
+ stepIndex: numberField(payload, "stepIndex") ?? context.stepIndex,
1035
+ kind: decisionKind,
1036
+ chosenAction,
1037
+ candidateActions: stringArrayField(payload, "candidateActions"),
1038
+ confidence: finiteUnitField(payload, "confidence"),
1039
+ behaviorProb: numberField(payload, "behaviorProb"),
1040
+ targetProb: numberField(payload, "targetProb"),
1041
+ qHatChosen: finiteUnitField(payload, "qHatChosen"),
1042
+ vHatTarget: finiteUnitField(payload, "vHatTarget"),
1043
+ costUsd: nonNegativeNumberField(payload, "costUsd"),
1044
+ evidence,
1045
+ outcome: parseOutcome(payload),
1046
+ metadata: recordField(payload, "metadata")
1047
+ } };
1085
1048
  }
1086
1049
  function parseOutcome(payload) {
1087
- const value = recordField(payload, "outcome");
1088
- if (!value) return void 0;
1089
- return {
1090
- success: typeof value.success === "boolean" ? value.success : void 0,
1091
- score: finiteUnitField(value, "score"),
1092
- reward: finiteUnitField(value, "reward"),
1093
- costUsd: nonNegativeNumberField(value, "costUsd"),
1094
- observedAt: stringField(value, "observedAt"),
1095
- metadata: recordField(value, "metadata")
1096
- };
1050
+ const value = recordField(payload, "outcome");
1051
+ if (!value) return void 0;
1052
+ return {
1053
+ success: typeof value.success === "boolean" ? value.success : void 0,
1054
+ score: finiteUnitField(value, "score"),
1055
+ reward: finiteUnitField(value, "reward"),
1056
+ costUsd: nonNegativeNumberField(value, "costUsd"),
1057
+ observedAt: stringField(value, "observedAt"),
1058
+ metadata: recordField(value, "metadata")
1059
+ };
1097
1060
  }
1098
1061
  function stringField(obj, key) {
1099
- const value = obj[key];
1100
- return typeof value === "string" && value.length > 0 ? value : void 0;
1062
+ const value = obj[key];
1063
+ return typeof value === "string" && value.length > 0 ? value : void 0;
1101
1064
  }
1102
1065
  function numberField(obj, key) {
1103
- const value = obj[key];
1104
- return typeof value === "number" && Number.isFinite(value) ? value : void 0;
1066
+ const value = obj[key];
1067
+ return typeof value === "number" && Number.isFinite(value) ? value : void 0;
1105
1068
  }
1106
1069
  function finiteUnitField(obj, key) {
1107
- const value = numberField(obj, key);
1108
- return value === void 0 ? void 0 : Math.max(0, Math.min(1, value));
1070
+ const value = numberField(obj, key);
1071
+ return value === void 0 ? void 0 : Math.max(0, Math.min(1, value));
1109
1072
  }
1110
1073
  function nonNegativeNumberField(obj, key) {
1111
- const value = numberField(obj, key);
1112
- return value === void 0 ? void 0 : Math.max(0, value);
1074
+ const value = numberField(obj, key);
1075
+ return value === void 0 ? void 0 : Math.max(0, value);
1113
1076
  }
1114
1077
  function stringArrayField(obj, key) {
1115
- const value = obj[key];
1116
- if (!Array.isArray(value)) return void 0;
1117
- const strings = value.filter(
1118
- (item) => typeof item === "string" && item.length > 0
1119
- );
1120
- return strings.length > 0 ? strings : void 0;
1078
+ const value = obj[key];
1079
+ if (!Array.isArray(value)) return void 0;
1080
+ const strings = value.filter((item) => typeof item === "string" && item.length > 0);
1081
+ return strings.length > 0 ? strings : void 0;
1121
1082
  }
1122
1083
  function recordField(obj, key) {
1123
- const value = obj[key];
1124
- if (!value || typeof value !== "object" || Array.isArray(value)) return void 0;
1125
- return value;
1126
- }
1127
-
1128
- // src/belief-state/runtime-hooks.ts
1129
- var DEFAULT_MAX_CONTEXT_CHARS = 12e3;
1130
- var DEFAULT_PAYLOAD_PREVIEW_CHARS = 2e3;
1084
+ const value = obj[key];
1085
+ if (!value || typeof value !== "object" || Array.isArray(value)) return void 0;
1086
+ return value;
1087
+ }
1088
+ //#endregion
1089
+ //#region src/belief-state/runtime-hooks.ts
1090
+ const DEFAULT_MAX_CONTEXT_CHARS$1 = 12e3;
1091
+ const DEFAULT_PAYLOAD_PREVIEW_CHARS = 2e3;
1131
1092
  function runtimeDecisionPointToBeliefShadowProbeInput(point, options) {
1132
- const diagnostics = [];
1133
- const decisionKind = resolveDecisionKind(point, options.decisionKind, diagnostics);
1134
- if (!decisionKind) return { diagnostics };
1135
- const lifecycleEvidence = runtimeHookEventsToEvidenceRefs(point, options);
1136
- const evidence = [...point.evidence ?? [], ...lifecycleEvidence];
1137
- return {
1138
- input: {
1139
- probeId: options.probeId,
1140
- decisionId: point.id,
1141
- runId: point.runId,
1142
- scenarioId: point.scenarioId,
1143
- stepIndex: point.stepIndex,
1144
- decisionKind,
1145
- candidateActions: uniqueStrings(point.candidateActions ?? []),
1146
- evidence: evidence.map((ref) => ({
1147
- id: ref.id,
1148
- source: ref.source,
1149
- ...options.includeEvidenceDetail && ref.detail ? { detail: ref.detail } : {},
1150
- ...ref.quality ? { quality: ref.quality } : {}
1151
- })),
1152
- context: trimText(point.context, options.maxContextChars),
1153
- metadata: mergeMetadata(point.metadata, lifecycleMetadata(lifecycleEvidence))
1154
- },
1155
- diagnostics
1156
- };
1093
+ const diagnostics = [];
1094
+ const decisionKind = resolveDecisionKind(point, options.decisionKind, diagnostics);
1095
+ if (!decisionKind) return { diagnostics };
1096
+ const lifecycleEvidence = runtimeHookEventsToEvidenceRefs(point, options);
1097
+ const evidence = [...point.evidence ?? [], ...lifecycleEvidence];
1098
+ return {
1099
+ input: {
1100
+ probeId: options.probeId,
1101
+ decisionId: point.id,
1102
+ runId: point.runId,
1103
+ scenarioId: point.scenarioId,
1104
+ stepIndex: point.stepIndex,
1105
+ decisionKind,
1106
+ candidateActions: uniqueStrings$1(point.candidateActions ?? []),
1107
+ evidence: evidence.map((ref) => ({
1108
+ id: ref.id,
1109
+ source: ref.source,
1110
+ ...options.includeEvidenceDetail && ref.detail ? { detail: ref.detail } : {},
1111
+ ...ref.quality ? { quality: ref.quality } : {}
1112
+ })),
1113
+ context: trimText$1(point.context, options.maxContextChars),
1114
+ metadata: mergeMetadata(point.metadata, lifecycleMetadata(lifecycleEvidence))
1115
+ },
1116
+ diagnostics
1117
+ };
1157
1118
  }
1158
1119
  function runtimeDecisionPointToBeliefDecisionPoint(point, options) {
1159
- const diagnostics = [];
1160
- const decisionKind = resolveDecisionKind(point, options.decisionKind, diagnostics);
1161
- const chosenAction = stringOrUndefined(options.chosenAction);
1162
- if (!chosenAction) {
1163
- diagnostics.push({
1164
- decisionId: point.id,
1165
- severity: "error",
1166
- reason: "missing chosenAction"
1167
- });
1168
- }
1169
- const candidateActions = uniqueStrings(point.candidateActions ?? []);
1170
- if (chosenAction && candidateActions.length > 0 && !candidateActions.includes(chosenAction)) {
1171
- diagnostics.push({
1172
- decisionId: point.id,
1173
- severity: "warning",
1174
- reason: `chosenAction ${chosenAction} is not in candidateActions`
1175
- });
1176
- }
1177
- if (!decisionKind || !chosenAction) return { diagnostics };
1178
- const lifecycleEvidence = runtimeHookEventsToEvidenceRefs(point, options);
1179
- const evidence = [...point.evidence ?? [], ...lifecycleEvidence];
1180
- return {
1181
- point: {
1182
- id: point.id,
1183
- runId: point.runId,
1184
- scenarioId: point.scenarioId,
1185
- stepIndex: point.stepIndex,
1186
- kind: decisionKind,
1187
- chosenAction,
1188
- candidateActions,
1189
- confidence: unitProbabilityOrUndefined(options.confidence),
1190
- behaviorProb: finiteNumberOrUndefined(options.behaviorProb),
1191
- targetProb: finiteNumberOrUndefined(options.targetProb),
1192
- qHatChosen: options.qHatChosen === null ? null : unitProbabilityOrUndefined(options.qHatChosen),
1193
- vHatTarget: options.vHatTarget === null ? null : unitProbabilityOrUndefined(options.vHatTarget),
1194
- qHat: options.qHat === null ? null : unitProbabilityOrUndefined(options.qHat),
1195
- costUsd: nonNegativeNumberOrUndefined(options.costUsd),
1196
- evidence: evidence.map((ref) => runtimeEvidenceToBeliefEvidence(ref, point)),
1197
- outcome: options.outcome,
1198
- metadata: mergeMetadata(
1199
- mergeMetadata(point.metadata, lifecycleMetadata(lifecycleEvidence)),
1200
- options.metadata
1201
- )
1202
- },
1203
- diagnostics
1204
- };
1120
+ const diagnostics = [];
1121
+ const decisionKind = resolveDecisionKind(point, options.decisionKind, diagnostics);
1122
+ const chosenAction = stringOrUndefined(options.chosenAction);
1123
+ if (!chosenAction) diagnostics.push({
1124
+ decisionId: point.id,
1125
+ severity: "error",
1126
+ reason: "missing chosenAction"
1127
+ });
1128
+ const candidateActions = uniqueStrings$1(point.candidateActions ?? []);
1129
+ if (chosenAction && candidateActions.length > 0 && !candidateActions.includes(chosenAction)) diagnostics.push({
1130
+ decisionId: point.id,
1131
+ severity: "warning",
1132
+ reason: `chosenAction ${chosenAction} is not in candidateActions`
1133
+ });
1134
+ if (!decisionKind || !chosenAction) return { diagnostics };
1135
+ const lifecycleEvidence = runtimeHookEventsToEvidenceRefs(point, options);
1136
+ const evidence = [...point.evidence ?? [], ...lifecycleEvidence];
1137
+ return {
1138
+ point: {
1139
+ id: point.id,
1140
+ runId: point.runId,
1141
+ scenarioId: point.scenarioId,
1142
+ stepIndex: point.stepIndex,
1143
+ kind: decisionKind,
1144
+ chosenAction,
1145
+ candidateActions,
1146
+ confidence: unitProbabilityOrUndefined(options.confidence),
1147
+ behaviorProb: finiteNumberOrUndefined(options.behaviorProb),
1148
+ targetProb: finiteNumberOrUndefined(options.targetProb),
1149
+ qHatChosen: options.qHatChosen === null ? null : unitProbabilityOrUndefined(options.qHatChosen),
1150
+ vHatTarget: options.vHatTarget === null ? null : unitProbabilityOrUndefined(options.vHatTarget),
1151
+ costUsd: nonNegativeNumberOrUndefined(options.costUsd),
1152
+ evidence: evidence.map((ref) => runtimeEvidenceToBeliefEvidence(ref, point)),
1153
+ outcome: options.outcome,
1154
+ metadata: mergeMetadata(mergeMetadata(point.metadata, lifecycleMetadata(lifecycleEvidence)), options.metadata)
1155
+ },
1156
+ diagnostics
1157
+ };
1205
1158
  }
1206
1159
  function createBeliefRuntimeHookCollector(defaults) {
1207
- const decisions = [];
1208
- const events = [];
1209
- return {
1210
- hooks: {
1211
- onEvent: (event) => {
1212
- events.push(snapshotRuntimeHookEvent(event));
1213
- },
1214
- onDecisionPoint: (point) => {
1215
- decisions.push(snapshotRuntimeDecisionPoint(point));
1216
- }
1217
- },
1218
- decisions,
1219
- events,
1220
- toShadowProbeInputs: (options = {}) => {
1221
- const inputs = [];
1222
- const diagnostics = [];
1223
- const includeLifecycleEvidence = options.includeLifecycleEvidence ?? defaults.includeLifecycleEvidence;
1224
- for (const point of decisions) {
1225
- const report = runtimeDecisionPointToBeliefShadowProbeInput(point, {
1226
- ...defaults,
1227
- ...options,
1228
- includeLifecycleEvidence,
1229
- lifecycleEvents: includeLifecycleEvidence === false ? void 0 : options.lifecycleEvents ?? defaults.lifecycleEvents ?? events
1230
- });
1231
- if (report.input) inputs.push(report.input);
1232
- diagnostics.push(...report.diagnostics);
1233
- }
1234
- return { inputs, diagnostics };
1235
- },
1236
- clear: () => {
1237
- decisions.length = 0;
1238
- events.length = 0;
1239
- }
1240
- };
1160
+ const decisions = [];
1161
+ const events = [];
1162
+ return {
1163
+ hooks: {
1164
+ onEvent: (event) => {
1165
+ events.push(snapshotRuntimeHookEvent(event));
1166
+ },
1167
+ onDecisionPoint: (point) => {
1168
+ decisions.push(snapshotRuntimeDecisionPoint(point));
1169
+ }
1170
+ },
1171
+ decisions,
1172
+ events,
1173
+ toShadowProbeInputs: (options = {}) => {
1174
+ const inputs = [];
1175
+ const diagnostics = [];
1176
+ const includeLifecycleEvidence = options.includeLifecycleEvidence ?? defaults.includeLifecycleEvidence;
1177
+ for (const point of decisions) {
1178
+ const report = runtimeDecisionPointToBeliefShadowProbeInput(point, {
1179
+ ...defaults,
1180
+ ...options,
1181
+ includeLifecycleEvidence,
1182
+ lifecycleEvents: includeLifecycleEvidence === false ? void 0 : options.lifecycleEvents ?? defaults.lifecycleEvents ?? events
1183
+ });
1184
+ if (report.input) inputs.push(report.input);
1185
+ diagnostics.push(...report.diagnostics);
1186
+ }
1187
+ return {
1188
+ inputs,
1189
+ diagnostics
1190
+ };
1191
+ },
1192
+ clear: () => {
1193
+ decisions.length = 0;
1194
+ events.length = 0;
1195
+ }
1196
+ };
1241
1197
  }
1242
1198
  function resolveDecisionKind(point, override, diagnostics) {
1243
- const kind = override ?? point.kind;
1244
- if (isBeliefDecisionKind(kind)) return kind;
1245
- diagnostics.push({
1246
- decisionId: point.id,
1247
- severity: "error",
1248
- reason: `unsupported decisionKind "${kind}"`
1249
- });
1250
- return void 0;
1199
+ const kind = override ?? point.kind;
1200
+ if (isBeliefDecisionKind(kind)) return kind;
1201
+ diagnostics.push({
1202
+ decisionId: point.id,
1203
+ severity: "error",
1204
+ reason: `unsupported decisionKind "${kind}"`
1205
+ });
1251
1206
  }
1252
1207
  function runtimeEvidenceToBeliefEvidence(ref, point) {
1253
- if (isBeliefEvidenceSource(ref.source)) {
1254
- return {
1255
- source: ref.source,
1256
- id: ref.id,
1257
- runId: point.runId,
1258
- detail: ref.detail,
1259
- quality: ref.quality,
1260
- metadata: ref.metadata
1261
- };
1262
- }
1263
- return {
1264
- source: "event",
1265
- id: ref.id,
1266
- runId: point.runId,
1267
- detail: ref.detail,
1268
- quality: ref.quality,
1269
- metadata: mergeMetadata({ runtimeSource: ref.source }, ref.metadata)
1270
- };
1208
+ if (isBeliefEvidenceSource(ref.source)) return {
1209
+ source: ref.source,
1210
+ id: ref.id,
1211
+ runId: point.runId,
1212
+ detail: ref.detail,
1213
+ quality: ref.quality,
1214
+ metadata: ref.metadata
1215
+ };
1216
+ return {
1217
+ source: "event",
1218
+ id: ref.id,
1219
+ runId: point.runId,
1220
+ detail: ref.detail,
1221
+ quality: ref.quality,
1222
+ metadata: mergeMetadata({ runtimeSource: ref.source }, ref.metadata)
1223
+ };
1271
1224
  }
1272
1225
  function runtimeHookEventsToEvidenceRefs(point, options) {
1273
- if (options.includeLifecycleEvidence === false) return [];
1274
- return (options.lifecycleEvents ?? []).filter((event) => runtimeHookEventMatchesDecision(point, event)).map(runtimeHookEventToEvidenceRef);
1226
+ if (options.includeLifecycleEvidence === false) return [];
1227
+ return (options.lifecycleEvents ?? []).filter((event) => runtimeHookEventMatchesDecision(point, event)).map(runtimeHookEventToEvidenceRef);
1275
1228
  }
1276
1229
  function runtimeHookEventMatchesDecision(point, event) {
1277
- if (event.runId !== point.runId) return false;
1278
- if (event.scenarioId && point.scenarioId && event.scenarioId !== point.scenarioId) return false;
1279
- return event.stepIndex === void 0 || event.stepIndex === point.stepIndex;
1230
+ if (event.runId !== point.runId) return false;
1231
+ if (event.scenarioId && point.scenarioId && event.scenarioId !== point.scenarioId) return false;
1232
+ return event.stepIndex === void 0 || event.stepIndex === point.stepIndex;
1280
1233
  }
1281
1234
  function runtimeHookEventToEvidenceRef(event) {
1282
- return {
1283
- source: "runtime_event",
1284
- id: event.id,
1285
- detail: `${event.target}:${event.phase}`,
1286
- quality: "direct",
1287
- metadata: mergeMetadata(
1288
- compactMetadata({
1289
- target: event.target,
1290
- phase: event.phase,
1291
- timestamp: event.timestamp,
1292
- stepIndex: event.stepIndex,
1293
- parentId: event.parentId,
1294
- payloadPreview: previewUnknown(event.payload)
1295
- }),
1296
- event.metadata
1297
- )
1298
- };
1235
+ return {
1236
+ source: "runtime_event",
1237
+ id: event.id,
1238
+ detail: `${event.target}:${event.phase}`,
1239
+ quality: "direct",
1240
+ metadata: mergeMetadata(compactMetadata$1({
1241
+ target: event.target,
1242
+ phase: event.phase,
1243
+ timestamp: event.timestamp,
1244
+ stepIndex: event.stepIndex,
1245
+ parentId: event.parentId,
1246
+ payloadPreview: previewUnknown(event.payload)
1247
+ }), event.metadata)
1248
+ };
1299
1249
  }
1300
1250
  function lifecycleMetadata(refs) {
1301
- if (refs.length === 0) return void 0;
1302
- return {
1303
- lifecycleEventCount: refs.length,
1304
- lifecycleEventIds: refs.map((ref) => ref.id)
1305
- };
1251
+ if (refs.length === 0) return void 0;
1252
+ return {
1253
+ lifecycleEventCount: refs.length,
1254
+ lifecycleEventIds: refs.map((ref) => ref.id)
1255
+ };
1306
1256
  }
1307
1257
  function snapshotRuntimeHookEvent(event) {
1308
- return {
1309
- id: event.id,
1310
- runId: event.runId,
1311
- scenarioId: event.scenarioId,
1312
- target: event.target,
1313
- phase: event.phase,
1314
- timestamp: event.timestamp,
1315
- stepIndex: event.stepIndex,
1316
- parentId: event.parentId,
1317
- payload: snapshotUnknown(event.payload),
1318
- metadata: event.metadata ? { ...event.metadata } : void 0
1319
- };
1258
+ return {
1259
+ id: event.id,
1260
+ runId: event.runId,
1261
+ scenarioId: event.scenarioId,
1262
+ target: event.target,
1263
+ phase: event.phase,
1264
+ timestamp: event.timestamp,
1265
+ stepIndex: event.stepIndex,
1266
+ parentId: event.parentId,
1267
+ payload: snapshotUnknown(event.payload),
1268
+ metadata: event.metadata ? { ...event.metadata } : void 0
1269
+ };
1320
1270
  }
1321
1271
  function snapshotRuntimeDecisionPoint(point) {
1322
- return {
1323
- id: point.id,
1324
- runId: point.runId,
1325
- scenarioId: point.scenarioId,
1326
- stepIndex: point.stepIndex,
1327
- kind: point.kind,
1328
- candidateActions: [...point.candidateActions ?? []],
1329
- context: point.context,
1330
- evidence: (point.evidence ?? []).map((ref) => ({
1331
- source: ref.source,
1332
- id: ref.id,
1333
- detail: ref.detail,
1334
- quality: ref.quality,
1335
- metadata: ref.metadata ? { ...ref.metadata } : void 0
1336
- })),
1337
- metadata: point.metadata ? { ...point.metadata } : void 0
1338
- };
1272
+ return {
1273
+ id: point.id,
1274
+ runId: point.runId,
1275
+ scenarioId: point.scenarioId,
1276
+ stepIndex: point.stepIndex,
1277
+ kind: point.kind,
1278
+ candidateActions: [...point.candidateActions ?? []],
1279
+ context: point.context,
1280
+ evidence: (point.evidence ?? []).map((ref) => ({
1281
+ source: ref.source,
1282
+ id: ref.id,
1283
+ detail: ref.detail,
1284
+ quality: ref.quality,
1285
+ metadata: ref.metadata ? { ...ref.metadata } : void 0
1286
+ })),
1287
+ metadata: point.metadata ? { ...point.metadata } : void 0
1288
+ };
1339
1289
  }
1340
1290
  function mergeMetadata(base, extra) {
1341
- if (!base && !extra) return void 0;
1342
- return { ...base ?? {}, ...extra ?? {} };
1291
+ if (!base && !extra) return void 0;
1292
+ return {
1293
+ ...base ?? {},
1294
+ ...extra ?? {}
1295
+ };
1343
1296
  }
1344
- function compactMetadata(values) {
1345
- const entries = Object.entries(values).filter(([, value]) => value !== void 0);
1346
- return entries.length > 0 ? Object.fromEntries(entries) : void 0;
1297
+ function compactMetadata$1(values) {
1298
+ const entries = Object.entries(values).filter(([, value]) => value !== void 0);
1299
+ return entries.length > 0 ? Object.fromEntries(entries) : void 0;
1347
1300
  }
1348
1301
  function previewUnknown(value, maxChars = DEFAULT_PAYLOAD_PREVIEW_CHARS) {
1349
- if (value === void 0) return void 0;
1350
- if (typeof value === "string") return trimText(value, maxChars);
1351
- try {
1352
- return trimText(JSON.stringify(value), maxChars);
1353
- } catch {
1354
- return trimText(String(value), maxChars);
1355
- }
1302
+ if (value === void 0) return void 0;
1303
+ if (typeof value === "string") return trimText$1(value, maxChars);
1304
+ try {
1305
+ return trimText$1(JSON.stringify(value), maxChars);
1306
+ } catch {
1307
+ return trimText$1(String(value), maxChars);
1308
+ }
1356
1309
  }
1357
1310
  function snapshotUnknown(value) {
1358
- if (Array.isArray(value)) return [...value];
1359
- if (isRecord2(value)) return { ...value };
1360
- return value;
1311
+ if (Array.isArray(value)) return [...value];
1312
+ if (isRecord$1(value)) return { ...value };
1313
+ return value;
1361
1314
  }
1362
- function isRecord2(value) {
1363
- return typeof value === "object" && value !== null && !Array.isArray(value);
1315
+ function isRecord$1(value) {
1316
+ return typeof value === "object" && value !== null && !Array.isArray(value);
1364
1317
  }
1365
- function uniqueStrings(values) {
1366
- return [...new Set(values.filter((value) => value.length > 0))];
1318
+ function uniqueStrings$1(values) {
1319
+ return [...new Set(values.filter((value) => value.length > 0))];
1367
1320
  }
1368
- function trimText(value, maxChars = DEFAULT_MAX_CONTEXT_CHARS) {
1369
- if (!value) return void 0;
1370
- return value.length > maxChars ? value.slice(value.length - maxChars) : value;
1321
+ function trimText$1(value, maxChars = DEFAULT_MAX_CONTEXT_CHARS$1) {
1322
+ if (!value) return void 0;
1323
+ return value.length > maxChars ? value.slice(value.length - maxChars) : value;
1371
1324
  }
1372
1325
  function stringOrUndefined(value) {
1373
- return typeof value === "string" && value.length > 0 ? value : void 0;
1326
+ return typeof value === "string" && value.length > 0 ? value : void 0;
1374
1327
  }
1375
1328
  function finiteNumberOrUndefined(value) {
1376
- return typeof value === "number" && Number.isFinite(value) ? value : void 0;
1329
+ return typeof value === "number" && Number.isFinite(value) ? value : void 0;
1377
1330
  }
1378
1331
  function unitProbabilityOrUndefined(value) {
1379
- const number = finiteNumberOrUndefined(value);
1380
- return number !== void 0 && number >= 0 && number <= 1 ? number : void 0;
1332
+ const number = finiteNumberOrUndefined(value);
1333
+ return number !== void 0 && number >= 0 && number <= 1 ? number : void 0;
1381
1334
  }
1382
1335
  function nonNegativeNumberOrUndefined(value) {
1383
- const number = finiteNumberOrUndefined(value);
1384
- return number !== void 0 && number >= 0 ? number : void 0;
1336
+ const number = finiteNumberOrUndefined(value);
1337
+ return number !== void 0 && number >= 0 ? number : void 0;
1385
1338
  }
1386
-
1387
- // src/belief-state/phase0-measurement.ts
1388
- var DEFAULT_BASELINE_POLICY_ID = "always-accept-observed-action";
1339
+ //#endregion
1340
+ //#region src/belief-state/phase0-measurement.ts
1341
+ const DEFAULT_BASELINE_POLICY_ID = "always-accept-observed-action";
1389
1342
  function buildRuntimeBeliefPhase0Measurement(options) {
1390
- const runsById = new Map(options.runs.map((run) => [run.runId, run]));
1391
- const labelsByDecisionId = /* @__PURE__ */ new Map();
1392
- const diagnostics = [];
1393
- for (const label of options.labels) {
1394
- if (labelsByDecisionId.has(label.decisionId)) {
1395
- diagnostics.push(`${label.decisionId}: duplicate label; using the last label`);
1396
- }
1397
- labelsByDecisionId.set(label.decisionId, label);
1398
- }
1399
- const points = [];
1400
- let missingRunRecordCount = 0;
1401
- let missingLabelCount = 0;
1402
- for (const decision of options.decisions) {
1403
- const run = runsById.get(decision.runId);
1404
- if (!run) {
1405
- missingRunRecordCount += 1;
1406
- diagnostics.push(`${decision.id}: missing RunRecord join for runId ${decision.runId}`);
1407
- continue;
1408
- }
1409
- const label = labelsByDecisionId.get(decision.id);
1410
- if (!label) {
1411
- missingLabelCount += 1;
1412
- diagnostics.push(`${decision.id}: missing observed action/outcome label`);
1413
- continue;
1414
- }
1415
- const splitTag = label.splitTag ?? run.splitTag;
1416
- const report = runtimeDecisionPointToBeliefDecisionPoint(
1417
- { ...decision, scenarioId: decision.scenarioId ?? run.scenarioId },
1418
- {
1419
- chosenAction: label.chosenAction,
1420
- confidence: label.confidence,
1421
- behaviorProb: label.behaviorProb,
1422
- targetProb: label.targetProb,
1423
- qHatChosen: label.qHatChosen,
1424
- vHatTarget: label.vHatTarget,
1425
- qHat: label.qHat,
1426
- costUsd: label.costUsd,
1427
- outcome: label.outcome,
1428
- lifecycleEvents: options.events,
1429
- metadata: compactMetadata2({
1430
- baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
1431
- splitTag,
1432
- ...label.metadata
1433
- })
1434
- }
1435
- );
1436
- diagnostics.push(...report.diagnostics.map((item) => `${item.decisionId}: ${item.reason}`));
1437
- if (report.point) points.push(report.point);
1438
- }
1439
- const packet = buildBeliefDecisionResearchEvidencePacket({
1440
- ...options,
1441
- points
1442
- });
1443
- return {
1444
- points,
1445
- packet,
1446
- summary: summarizePhase0Measurement(options, points, packet, {
1447
- missingRunRecordCount,
1448
- missingLabelCount
1449
- }),
1450
- diagnostics
1451
- };
1343
+ const runsById = new Map(options.runs.map((run) => [run.runId, run]));
1344
+ const labelsByDecisionId = /* @__PURE__ */ new Map();
1345
+ const diagnostics = [];
1346
+ for (const label of options.labels) {
1347
+ if (labelsByDecisionId.has(label.decisionId)) diagnostics.push(`${label.decisionId}: duplicate label; using the last label`);
1348
+ labelsByDecisionId.set(label.decisionId, label);
1349
+ }
1350
+ const points = [];
1351
+ let missingRunRecordCount = 0;
1352
+ let missingLabelCount = 0;
1353
+ for (const decision of options.decisions) {
1354
+ const run = runsById.get(decision.runId);
1355
+ if (!run) {
1356
+ missingRunRecordCount += 1;
1357
+ diagnostics.push(`${decision.id}: missing RunRecord join for runId ${decision.runId}`);
1358
+ continue;
1359
+ }
1360
+ const label = labelsByDecisionId.get(decision.id);
1361
+ if (!label) {
1362
+ missingLabelCount += 1;
1363
+ diagnostics.push(`${decision.id}: missing observed action/outcome label`);
1364
+ continue;
1365
+ }
1366
+ const splitTag = label.splitTag ?? run.splitTag;
1367
+ const report = runtimeDecisionPointToBeliefDecisionPoint({
1368
+ ...decision,
1369
+ scenarioId: decision.scenarioId ?? run.scenarioId
1370
+ }, {
1371
+ chosenAction: label.chosenAction,
1372
+ confidence: label.confidence,
1373
+ behaviorProb: label.behaviorProb,
1374
+ targetProb: label.targetProb,
1375
+ qHatChosen: label.qHatChosen,
1376
+ vHatTarget: label.vHatTarget,
1377
+ costUsd: label.costUsd,
1378
+ outcome: label.outcome,
1379
+ lifecycleEvents: options.events,
1380
+ metadata: compactMetadata({
1381
+ baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
1382
+ splitTag,
1383
+ ...label.metadata
1384
+ })
1385
+ });
1386
+ diagnostics.push(...report.diagnostics.map((item) => `${item.decisionId}: ${item.reason}`));
1387
+ if (report.point) points.push(report.point);
1388
+ }
1389
+ const packet = buildBeliefDecisionResearchEvidencePacket({
1390
+ ...options,
1391
+ points
1392
+ });
1393
+ return {
1394
+ points,
1395
+ packet,
1396
+ summary: summarizePhase0Measurement(options, points, packet, {
1397
+ missingRunRecordCount,
1398
+ missingLabelCount
1399
+ }),
1400
+ diagnostics
1401
+ };
1452
1402
  }
1453
1403
  function summarizePhase0Measurement(options, points, packet, counts) {
1454
- const producerDecisionCount = options.decisions.length;
1455
- return {
1456
- runCount: options.runs.length,
1457
- producerDecisionCount,
1458
- lifecycleEventCount: options.events?.length ?? 0,
1459
- labelCount: options.labels.length,
1460
- completedPointCount: points.length,
1461
- runJoinRate: ratio(producerDecisionCount - counts.missingRunRecordCount, producerDecisionCount),
1462
- labelJoinRate: ratio(points.length, producerDecisionCount),
1463
- missingRunRecordCount: counts.missingRunRecordCount,
1464
- missingLabelCount: counts.missingLabelCount,
1465
- withEvidence: points.filter((point) => point.evidence.length > 0).length,
1466
- withOutcome: points.filter((point) => point.outcome).length,
1467
- withSplit: points.filter((point) => typeof point.metadata?.splitTag === "string").length,
1468
- withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
1469
- withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
1470
- baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
1471
- packetStatus: packet.status,
1472
- claimScope: packet.claimScope
1473
- };
1404
+ const producerDecisionCount = options.decisions.length;
1405
+ return {
1406
+ runCount: options.runs.length,
1407
+ producerDecisionCount,
1408
+ lifecycleEventCount: options.events?.length ?? 0,
1409
+ labelCount: options.labels.length,
1410
+ completedPointCount: points.length,
1411
+ runJoinRate: ratio(producerDecisionCount - counts.missingRunRecordCount, producerDecisionCount),
1412
+ labelJoinRate: ratio(points.length, producerDecisionCount),
1413
+ missingRunRecordCount: counts.missingRunRecordCount,
1414
+ missingLabelCount: counts.missingLabelCount,
1415
+ withEvidence: points.filter((point) => point.evidence.length > 0).length,
1416
+ withOutcome: points.filter((point) => point.outcome).length,
1417
+ withSplit: points.filter((point) => typeof point.metadata?.splitTag === "string").length,
1418
+ withBehaviorProb: points.filter((point) => point.behaviorProb !== void 0).length,
1419
+ withTargetProb: points.filter((point) => point.targetProb !== void 0).length,
1420
+ baselinePolicyId: options.baselinePolicyId ?? DEFAULT_BASELINE_POLICY_ID,
1421
+ packetStatus: packet.status,
1422
+ claimScope: packet.claimScope
1423
+ };
1474
1424
  }
1475
1425
  function ratio(numerator, denominator) {
1476
- return denominator > 0 ? numerator / denominator : 0;
1426
+ return denominator > 0 ? numerator / denominator : 0;
1477
1427
  }
1478
- function compactMetadata2(values) {
1479
- const entries = Object.entries(values).filter(([, value]) => value !== void 0);
1480
- return entries.length > 0 ? Object.fromEntries(entries) : void 0;
1481
- }
1482
-
1483
- // src/belief-state/runtime-benchmark-corpus.ts
1484
- var MAX_STRING_LENGTH = 12e3;
1485
- var MAX_CONTEXT_LENGTH = 2e4;
1486
- var MAX_EVIDENCE_DETAIL_LENGTH = 2e3;
1487
- var MAX_CANDIDATE_ACTIONS = 50;
1488
- var MAX_EVIDENCE_REFS = 50;
1489
- var MAX_METADATA_DEPTH = 4;
1490
- var MAX_METADATA_KEYS = 100;
1491
- var SENSITIVE_KEY_RE = /(?:authorization|api[_-]?key|token|secret|password|cookie|credential|bearer)/i;
1492
- var SENSITIVE_VALUE_RES = [
1493
- /\bBearer\s+[A-Za-z0-9._~+/=-]+/gi,
1494
- /\b(?:sk|gh[pousr])_[A-Za-z0-9_]{20,}\b/g,
1495
- /\b(?:sk|ghp|gho|ghu|ghs|ghr)-[A-Za-z0-9_-]{20,}\b/g
1428
+ function compactMetadata(values) {
1429
+ const entries = Object.entries(values).filter(([, value]) => value !== void 0);
1430
+ return entries.length > 0 ? Object.fromEntries(entries) : void 0;
1431
+ }
1432
+ //#endregion
1433
+ //#region src/belief-state/runtime-benchmark-corpus.ts
1434
+ const MAX_STRING_LENGTH = 12e3;
1435
+ const MAX_CONTEXT_LENGTH = 2e4;
1436
+ const MAX_EVIDENCE_DETAIL_LENGTH = 2e3;
1437
+ const MAX_CANDIDATE_ACTIONS = 50;
1438
+ const MAX_EVIDENCE_REFS = 50;
1439
+ const MAX_METADATA_DEPTH = 4;
1440
+ const MAX_METADATA_KEYS = 100;
1441
+ const SENSITIVE_KEY_RE = /(?:authorization|api[_-]?key|token|secret|password|cookie|credential|bearer)/i;
1442
+ const SENSITIVE_VALUE_RES = [
1443
+ /\bBearer\s+[A-Za-z0-9._~+/=-]+/gi,
1444
+ /\b(?:sk|gh[pousr])_[A-Za-z0-9_]{20,}\b/g,
1445
+ /\b(?:sk|ghp|gho|ghu|ghs|ghr)-[A-Za-z0-9_-]{20,}\b/g
1496
1446
  ];
1497
- var SENSITIVE_ASSIGNMENT_RE = /\b(api[_-]?key|token|secret|password|cookie)\s*[:=]\s*["']?[^"'\s,;}]+/gi;
1447
+ const SENSITIVE_ASSIGNMENT_RE = /\b(api[_-]?key|token|secret|password|cookie)\s*[:=]\s*["']?[^"'\s,;}]+/gi;
1498
1448
  function buildRuntimeBenchmarkBeliefPhase0Measurement(options) {
1499
- const diagnostics = [];
1500
- const trajectory = projectRuntimeTrajectoryEvidence({
1501
- records: options.records,
1502
- defaultSplitTag: options.defaultSplitTag,
1503
- recordIdOf: runtimeBenchmarkRecordId,
1504
- scenarioIdOf: runtimeBenchmarkScenarioId
1505
- });
1506
- const decisions = options.decisions ?? runtimeBenchmarkDecisionPoints(options.records, diagnostics);
1507
- const labels = options.labels ?? [];
1508
- if (decisions.length === 0) {
1509
- diagnostics.push(
1510
- "no runtime decision points supplied or found on records; benchmark lifecycle events alone cannot produce belief decision rows"
1511
- );
1512
- }
1513
- if (labels.length === 0 && decisions.length > 0) {
1514
- diagnostics.push(
1515
- "no decision labels supplied; observed action/outcome joins will be incomplete"
1516
- );
1517
- }
1518
- const measurement = buildRuntimeBeliefPhase0Measurement({
1519
- ...options,
1520
- runs: trajectory.runs,
1521
- events: trajectory.events,
1522
- decisions,
1523
- labels
1524
- });
1525
- return {
1526
- runs: trajectory.runs,
1527
- events: trajectory.events,
1528
- decisions,
1529
- labels,
1530
- trajectory,
1531
- measurement,
1532
- summary: {
1533
- decisionCount: decisions.length,
1534
- labelCount: labels.length
1535
- },
1536
- diagnostics: [...trajectory.diagnostics, ...diagnostics, ...measurement.diagnostics]
1537
- };
1449
+ const diagnostics = [];
1450
+ const trajectory = projectRuntimeTrajectoryEvidence({
1451
+ records: options.records,
1452
+ defaultSplitTag: options.defaultSplitTag,
1453
+ recordIdOf: runtimeBenchmarkRecordId,
1454
+ scenarioIdOf: runtimeBenchmarkScenarioId
1455
+ });
1456
+ const decisions = options.decisions ?? runtimeBenchmarkDecisionPoints(options.records, diagnostics);
1457
+ const labels = options.labels ?? [];
1458
+ if (decisions.length === 0) diagnostics.push("no runtime decision points supplied or found on records; benchmark lifecycle events alone cannot produce belief decision rows");
1459
+ if (labels.length === 0 && decisions.length > 0) diagnostics.push("no decision labels supplied; observed action/outcome joins will be incomplete");
1460
+ const measurement = buildRuntimeBeliefPhase0Measurement({
1461
+ ...options,
1462
+ runs: trajectory.runs,
1463
+ events: trajectory.events,
1464
+ decisions,
1465
+ labels
1466
+ });
1467
+ return {
1468
+ runs: trajectory.runs,
1469
+ events: trajectory.events,
1470
+ decisions,
1471
+ labels,
1472
+ trajectory,
1473
+ measurement,
1474
+ summary: {
1475
+ decisionCount: decisions.length,
1476
+ labelCount: labels.length
1477
+ },
1478
+ diagnostics: [
1479
+ ...trajectory.diagnostics,
1480
+ ...diagnostics,
1481
+ ...measurement.diagnostics
1482
+ ]
1483
+ };
1538
1484
  }
1539
1485
  function runtimeBenchmarkRecordId(record) {
1540
- const parts = [
1541
- nonEmptyString(record.benchmark),
1542
- nonEmptyString(record.instanceId),
1543
- nonEmptyString(record.condition)
1544
- ].filter((part) => part !== void 0);
1545
- return parts.length > 0 ? parts.join(":") : void 0;
1486
+ const parts = [
1487
+ nonEmptyString(record.benchmark),
1488
+ nonEmptyString(record.instanceId),
1489
+ nonEmptyString(record.condition)
1490
+ ].filter((part) => part !== void 0);
1491
+ return parts.length > 0 ? parts.join(":") : void 0;
1546
1492
  }
1547
1493
  function runtimeBenchmarkScenarioId(record) {
1548
- return nonEmptyString(record.instanceId);
1494
+ return nonEmptyString(record.instanceId);
1549
1495
  }
1550
1496
  function runtimeBenchmarkDecisionPoints(records, diagnostics) {
1551
- const decisions = [];
1552
- for (let recordIndex = 0; recordIndex < records.length; recordIndex += 1) {
1553
- const record = records[recordIndex];
1554
- const raw = record.runtimeDecisionPoints;
1555
- if (raw === void 0) continue;
1556
- const recordId = runtimeBenchmarkRecordId(record) ?? `record[${recordIndex}]`;
1557
- if (!Array.isArray(raw)) {
1558
- diagnostics.push(`${recordId}: runtimeDecisionPoints is not an array`);
1559
- continue;
1560
- }
1561
- for (let pointIndex = 0; pointIndex < raw.length; pointIndex += 1) {
1562
- const point = runtimeBenchmarkDecisionPoint(raw[pointIndex], {
1563
- diagnostics,
1564
- path: `${recordId}: runtimeDecisionPoints[${pointIndex}]`
1565
- });
1566
- if (!point) {
1567
- diagnostics.push(
1568
- `${recordId}: runtimeDecisionPoints[${pointIndex}] is not a RuntimeDecisionPoint`
1569
- );
1570
- continue;
1571
- }
1572
- decisions.push(point);
1573
- }
1574
- }
1575
- return decisions;
1497
+ const decisions = [];
1498
+ for (let recordIndex = 0; recordIndex < records.length; recordIndex += 1) {
1499
+ const record = records[recordIndex];
1500
+ const raw = record.runtimeDecisionPoints;
1501
+ if (raw === void 0) continue;
1502
+ const recordId = runtimeBenchmarkRecordId(record) ?? `record[${recordIndex}]`;
1503
+ if (!Array.isArray(raw)) {
1504
+ diagnostics.push(`${recordId}: runtimeDecisionPoints is not an array`);
1505
+ continue;
1506
+ }
1507
+ for (let pointIndex = 0; pointIndex < raw.length; pointIndex += 1) {
1508
+ const point = runtimeBenchmarkDecisionPoint(raw[pointIndex], {
1509
+ diagnostics,
1510
+ path: `${recordId}: runtimeDecisionPoints[${pointIndex}]`
1511
+ });
1512
+ if (!point) {
1513
+ diagnostics.push(`${recordId}: runtimeDecisionPoints[${pointIndex}] is not a RuntimeDecisionPoint`);
1514
+ continue;
1515
+ }
1516
+ decisions.push(point);
1517
+ }
1518
+ }
1519
+ return decisions;
1576
1520
  }
1577
1521
  function runtimeBenchmarkDecisionPoint(input, context) {
1578
- if (!isRecord3(input)) return null;
1579
- if (typeof input.id !== "string" || input.id.length === 0) return null;
1580
- if (typeof input.runId !== "string" || input.runId.length === 0) return null;
1581
- if (typeof input.stepIndex !== "number" || !Number.isInteger(input.stepIndex) || input.stepIndex < 0) {
1582
- return null;
1583
- }
1584
- if (typeof input.kind !== "string" || input.kind.length === 0) return null;
1585
- return {
1586
- id: sanitizeString(input.id, MAX_STRING_LENGTH),
1587
- runId: sanitizeString(input.runId, MAX_STRING_LENGTH),
1588
- scenarioId: sanitizeOptionalString(input.scenarioId, MAX_STRING_LENGTH),
1589
- stepIndex: input.stepIndex,
1590
- kind: sanitizeString(input.kind, MAX_STRING_LENGTH),
1591
- candidateActions: stringArray(input.candidateActions, {
1592
- ...context,
1593
- maxItems: MAX_CANDIDATE_ACTIONS,
1594
- label: "candidateActions"
1595
- }),
1596
- context: sanitizeOptionalString(input.context, MAX_CONTEXT_LENGTH),
1597
- evidence: runtimeBenchmarkEvidence(input.evidence, context),
1598
- metadata: sanitizeMetadataRecord(input.metadata)
1599
- };
1522
+ if (!isRecord(input)) return null;
1523
+ if (typeof input.id !== "string" || input.id.length === 0) return null;
1524
+ if (typeof input.runId !== "string" || input.runId.length === 0) return null;
1525
+ if (typeof input.stepIndex !== "number" || !Number.isInteger(input.stepIndex) || input.stepIndex < 0) return null;
1526
+ if (typeof input.kind !== "string" || input.kind.length === 0) return null;
1527
+ return {
1528
+ id: sanitizeString(input.id, MAX_STRING_LENGTH),
1529
+ runId: sanitizeString(input.runId, MAX_STRING_LENGTH),
1530
+ scenarioId: sanitizeOptionalString(input.scenarioId, MAX_STRING_LENGTH),
1531
+ stepIndex: input.stepIndex,
1532
+ kind: sanitizeString(input.kind, MAX_STRING_LENGTH),
1533
+ candidateActions: stringArray(input.candidateActions, {
1534
+ ...context,
1535
+ maxItems: MAX_CANDIDATE_ACTIONS,
1536
+ label: "candidateActions"
1537
+ }),
1538
+ context: sanitizeOptionalString(input.context, MAX_CONTEXT_LENGTH),
1539
+ evidence: runtimeBenchmarkEvidence(input.evidence, context),
1540
+ metadata: sanitizeMetadataRecord(input.metadata)
1541
+ };
1600
1542
  }
1601
1543
  function runtimeBenchmarkEvidence(input, context) {
1602
- if (!Array.isArray(input)) return [];
1603
- if (input.length > MAX_EVIDENCE_REFS) {
1604
- context.diagnostics.push(`${context.path}: evidence truncated to ${MAX_EVIDENCE_REFS} refs`);
1605
- }
1606
- return input.slice(0, MAX_EVIDENCE_REFS).flatMap((item) => {
1607
- if (!isRecord3(item)) return [];
1608
- const source = sanitizeOptionalString(item.source, MAX_STRING_LENGTH);
1609
- const id = sanitizeOptionalString(item.id, MAX_STRING_LENGTH);
1610
- if (!source || !id) return [];
1611
- return [
1612
- {
1613
- source,
1614
- id,
1615
- detail: sanitizeOptionalString(item.detail, MAX_EVIDENCE_DETAIL_LENGTH),
1616
- metadata: sanitizeMetadataRecord(item.metadata)
1617
- }
1618
- ];
1619
- });
1544
+ if (!Array.isArray(input)) return [];
1545
+ if (input.length > MAX_EVIDENCE_REFS) context.diagnostics.push(`${context.path}: evidence truncated to ${MAX_EVIDENCE_REFS} refs`);
1546
+ return input.slice(0, MAX_EVIDENCE_REFS).flatMap((item) => {
1547
+ if (!isRecord(item)) return [];
1548
+ const source = sanitizeOptionalString(item.source, MAX_STRING_LENGTH);
1549
+ const id = sanitizeOptionalString(item.id, MAX_STRING_LENGTH);
1550
+ if (!source || !id) return [];
1551
+ return [{
1552
+ source,
1553
+ id,
1554
+ detail: sanitizeOptionalString(item.detail, MAX_EVIDENCE_DETAIL_LENGTH),
1555
+ metadata: sanitizeMetadataRecord(item.metadata)
1556
+ }];
1557
+ });
1620
1558
  }
1621
1559
  function stringArray(input, context) {
1622
- if (!Array.isArray(input)) return void 0;
1623
- if (input.length > context.maxItems) {
1624
- context.diagnostics.push(`${context.path}: ${context.label} truncated to ${context.maxItems}`);
1625
- }
1626
- const values = input.slice(0, context.maxItems).filter((value) => typeof value === "string" && value.length > 0).map((value) => sanitizeString(value, MAX_STRING_LENGTH));
1627
- return values.length > 0 ? values : void 0;
1560
+ if (!Array.isArray(input)) return void 0;
1561
+ if (input.length > context.maxItems) context.diagnostics.push(`${context.path}: ${context.label} truncated to ${context.maxItems}`);
1562
+ const values = input.slice(0, context.maxItems).filter((value) => typeof value === "string" && value.length > 0).map((value) => sanitizeString(value, MAX_STRING_LENGTH));
1563
+ return values.length > 0 ? values : void 0;
1628
1564
  }
1629
1565
  function sanitizeMetadataRecord(metadata) {
1630
- if (!isRecord3(metadata)) return void 0;
1631
- const sanitized = sanitizeMetadata(metadata);
1632
- if (!sanitized || typeof sanitized !== "object" || Array.isArray(sanitized)) return void 0;
1633
- return sanitized;
1566
+ if (!isRecord(metadata)) return void 0;
1567
+ const sanitized = sanitizeMetadata(metadata);
1568
+ if (!sanitized || typeof sanitized !== "object" || Array.isArray(sanitized)) return void 0;
1569
+ return sanitized;
1634
1570
  }
1635
1571
  function sanitizeMetadata(value, depth = 0) {
1636
- if (value == null) return value;
1637
- if (typeof value === "string") return sanitizeString(value, MAX_STRING_LENGTH);
1638
- if (typeof value === "number" || typeof value === "boolean") return value;
1639
- if (Array.isArray(value)) {
1640
- if (depth >= MAX_METADATA_DEPTH) return "[MaxDepth]";
1641
- return value.slice(0, MAX_METADATA_KEYS).map((item) => sanitizeMetadata(item, depth + 1));
1642
- }
1643
- if (!isRecord3(value)) return void 0;
1644
- if (depth >= MAX_METADATA_DEPTH) return "[MaxDepth]";
1645
- const sanitized = {};
1646
- for (const [key, nested] of Object.entries(value).slice(0, MAX_METADATA_KEYS)) {
1647
- sanitized[key] = SENSITIVE_KEY_RE.test(key) ? "[REDACTED]" : sanitizeMetadata(nested, depth + 1);
1648
- }
1649
- return sanitized;
1572
+ if (value == null) return value;
1573
+ if (typeof value === "string") return sanitizeString(value, MAX_STRING_LENGTH);
1574
+ if (typeof value === "number" || typeof value === "boolean") return value;
1575
+ if (Array.isArray(value)) {
1576
+ if (depth >= MAX_METADATA_DEPTH) return "[MaxDepth]";
1577
+ return value.slice(0, MAX_METADATA_KEYS).map((item) => sanitizeMetadata(item, depth + 1));
1578
+ }
1579
+ if (!isRecord(value)) return void 0;
1580
+ if (depth >= MAX_METADATA_DEPTH) return "[MaxDepth]";
1581
+ const sanitized = {};
1582
+ for (const [key, nested] of Object.entries(value).slice(0, MAX_METADATA_KEYS)) sanitized[key] = SENSITIVE_KEY_RE.test(key) ? "[REDACTED]" : sanitizeMetadata(nested, depth + 1);
1583
+ return sanitized;
1650
1584
  }
1651
1585
  function sanitizeOptionalString(value, maxLength) {
1652
- return typeof value === "string" && value.length > 0 ? sanitizeString(value, maxLength) : void 0;
1586
+ return typeof value === "string" && value.length > 0 ? sanitizeString(value, maxLength) : void 0;
1653
1587
  }
1654
1588
  function sanitizeString(value, maxLength) {
1655
- let sanitized = value;
1656
- for (const pattern of SENSITIVE_VALUE_RES) {
1657
- sanitized = sanitized.replace(pattern, "[REDACTED]");
1658
- }
1659
- sanitized = sanitized.replace(
1660
- SENSITIVE_ASSIGNMENT_RE,
1661
- (_match, key) => `${key}=[REDACTED]`
1662
- );
1663
- if (sanitized.length <= maxLength) return sanitized;
1664
- return sanitized.slice(0, maxLength);
1665
- }
1666
- function isRecord3(value) {
1667
- return typeof value === "object" && value !== null && !Array.isArray(value);
1589
+ let sanitized = value;
1590
+ for (const pattern of SENSITIVE_VALUE_RES) sanitized = sanitized.replace(pattern, "[REDACTED]");
1591
+ sanitized = sanitized.replace(SENSITIVE_ASSIGNMENT_RE, (_match, key) => `${key}=[REDACTED]`);
1592
+ if (sanitized.length <= maxLength) return sanitized;
1593
+ return sanitized.slice(0, maxLength);
1594
+ }
1595
+ function isRecord(value) {
1596
+ return typeof value === "object" && value !== null && !Array.isArray(value);
1668
1597
  }
1669
1598
  function nonEmptyString(value) {
1670
- return typeof value === "string" && value.length > 0 ? value : void 0;
1599
+ return typeof value === "string" && value.length > 0 ? value : void 0;
1671
1600
  }
1672
-
1673
- // src/belief-state/shadow-probe.ts
1674
- var DEFAULT_CONCURRENCY = 4;
1675
- var DEFAULT_MAX_CONTEXT_CHARS2 = 12e3;
1601
+ //#endregion
1602
+ //#region src/belief-state/shadow-probe.ts
1603
+ const DEFAULT_CONCURRENCY = 4;
1604
+ const DEFAULT_MAX_CONTEXT_CHARS = 12e3;
1676
1605
  async function runBeliefShadowProbe(options) {
1677
- const concurrency = boundedInteger(options.concurrency ?? DEFAULT_CONCURRENCY, 1, 32);
1678
- const records = [];
1679
- const diagnostics = [];
1680
- let next = 0;
1681
- async function worker() {
1682
- while (next < options.points.length) {
1683
- const index = next;
1684
- next += 1;
1685
- const point = options.points[index];
1686
- if (!point) continue;
1687
- const result = await probePoint(point, options);
1688
- records[index] = result.record;
1689
- diagnostics.push(...result.diagnostics);
1690
- }
1691
- }
1692
- await Promise.all(Array.from({ length: Math.min(concurrency, options.points.length) }, worker));
1693
- const completed = records.filter((record) => !!record);
1694
- return {
1695
- probeId: options.probeId,
1696
- records: completed,
1697
- diagnostics,
1698
- summary: summarizeShadowProbe(options.points.length, completed)
1699
- };
1606
+ const concurrency = boundedInteger(options.concurrency ?? DEFAULT_CONCURRENCY, 1, 32);
1607
+ const records = [];
1608
+ const diagnostics = [];
1609
+ let next = 0;
1610
+ async function worker() {
1611
+ while (next < options.points.length) {
1612
+ const index = next;
1613
+ next += 1;
1614
+ const point = options.points[index];
1615
+ if (!point) continue;
1616
+ const result = await probePoint(point, options);
1617
+ records[index] = result.record;
1618
+ diagnostics.push(...result.diagnostics);
1619
+ }
1620
+ }
1621
+ await Promise.all(Array.from({ length: Math.min(concurrency, options.points.length) }, worker));
1622
+ const completed = records.filter((record) => !!record);
1623
+ return {
1624
+ probeId: options.probeId,
1625
+ records: completed,
1626
+ diagnostics,
1627
+ summary: summarizeShadowProbe(options.points.length, completed)
1628
+ };
1700
1629
  }
1701
1630
  function formatBeliefShadowProbePrompt(input) {
1702
- return [
1703
- "Return only JSON. Do not include chain-of-thought.",
1704
- "Infer the agent belief state at this decision boundary using only the context below.",
1705
- "",
1706
- `decisionKind: ${input.decisionKind}`,
1707
- `candidateActions: ${JSON.stringify(input.candidateActions)}`,
1708
- input.observedAction ? `observedAction: ${JSON.stringify(input.observedAction)}` : "",
1709
- input.context ? `context:
1710
- ${input.context}` : "",
1711
- "",
1712
- "Schema:",
1713
- JSON.stringify({
1714
- predictedAction: "one candidate action",
1715
- confidence: "number in [0,1]",
1716
- beliefSummary: "short outcome-blind summary",
1717
- uncertainty: ["short uncertainty"],
1718
- evidenceRefs: ["evidence id"],
1719
- wouldChangeMindIf: ["observable evidence"],
1720
- targetProb: "optional number in [0,1]",
1721
- qHat: "optional number in [0,1]"
1722
- })
1723
- ].filter(Boolean).join("\n");
1631
+ return [
1632
+ "Return only JSON. Do not include chain-of-thought.",
1633
+ "Infer the agent belief state at this decision boundary using only the context below.",
1634
+ "",
1635
+ `decisionKind: ${input.decisionKind}`,
1636
+ `candidateActions: ${JSON.stringify(input.candidateActions)}`,
1637
+ input.observedAction ? `observedAction: ${JSON.stringify(input.observedAction)}` : "",
1638
+ input.context ? `context:\n${input.context}` : "",
1639
+ "",
1640
+ "Schema:",
1641
+ JSON.stringify({
1642
+ predictedAction: "one candidate action",
1643
+ confidence: "number in [0,1]",
1644
+ beliefSummary: "short outcome-blind summary",
1645
+ uncertainty: ["short uncertainty"],
1646
+ evidenceRefs: ["evidence id"],
1647
+ wouldChangeMindIf: ["observable evidence"],
1648
+ targetProb: "optional number in [0,1]",
1649
+ qHatChosen: "optional number in [0,1], paired with vHatTarget",
1650
+ vHatTarget: "optional number in [0,1], paired with qHatChosen"
1651
+ })
1652
+ ].filter(Boolean).join("\n");
1724
1653
  }
1725
1654
  async function probePoint(point, options) {
1726
- const diagnostics = [];
1727
- const candidateActions = uniqueStrings2(point.candidateActions ?? []);
1728
- if ((options.requireCandidateActions ?? true) && candidateActions.length === 0) {
1729
- diagnostics.push({
1730
- decisionId: point.id,
1731
- severity: "warning",
1732
- reason: "missing candidateActions"
1733
- });
1734
- return { diagnostics };
1735
- }
1736
- let response;
1737
- try {
1738
- response = await options.probe({
1739
- probeId: options.probeId,
1740
- decisionId: point.id,
1741
- runId: point.runId,
1742
- scenarioId: point.scenarioId,
1743
- stepIndex: point.stepIndex,
1744
- decisionKind: point.kind,
1745
- candidateActions,
1746
- ...options.includeObservedAction ? { observedAction: point.chosenAction } : {},
1747
- evidence: point.evidence.map((ref) => ({
1748
- id: ref.id,
1749
- source: ref.source,
1750
- ...options.includeEvidenceDetail && ref.detail ? { detail: ref.detail } : {},
1751
- ...ref.quality ? { quality: ref.quality } : {}
1752
- })),
1753
- context: trimText2(await options.contextOf?.(point), options.maxContextChars),
1754
- metadata: await options.metadataOf?.(point)
1755
- });
1756
- } catch (error) {
1757
- diagnostics.push({
1758
- decisionId: point.id,
1759
- severity: "error",
1760
- reason: `probe threw: ${errorMessage2(error)}`
1761
- });
1762
- return { diagnostics };
1763
- }
1764
- const normalized = normalizeProbeResponse(response, {
1765
- point,
1766
- candidateActions,
1767
- allowOutOfSetActions: options.allowOutOfSetActions ?? false
1768
- });
1769
- if (!normalized.record) {
1770
- diagnostics.push(...normalized.diagnostics);
1771
- return { diagnostics };
1772
- }
1773
- return {
1774
- record: {
1775
- probeId: options.probeId,
1776
- decisionId: point.id,
1777
- runId: point.runId,
1778
- scenarioId: point.scenarioId,
1779
- stepIndex: point.stepIndex,
1780
- decisionKind: point.kind,
1781
- candidateActions,
1782
- observedAction: point.chosenAction,
1783
- agreesWithObservedAction: normalized.record.predictedAction === point.chosenAction,
1784
- ...options.includeOutcomeInRecord === false ? {} : { outcome: point.outcome },
1785
- ...normalized.record
1786
- },
1787
- diagnostics
1788
- };
1655
+ const diagnostics = [];
1656
+ const candidateActions = uniqueStrings(point.candidateActions ?? []);
1657
+ if ((options.requireCandidateActions ?? true) && candidateActions.length === 0) {
1658
+ diagnostics.push({
1659
+ decisionId: point.id,
1660
+ severity: "warning",
1661
+ reason: "missing candidateActions"
1662
+ });
1663
+ return { diagnostics };
1664
+ }
1665
+ let response;
1666
+ try {
1667
+ response = await options.probe({
1668
+ probeId: options.probeId,
1669
+ decisionId: point.id,
1670
+ runId: point.runId,
1671
+ scenarioId: point.scenarioId,
1672
+ stepIndex: point.stepIndex,
1673
+ decisionKind: point.kind,
1674
+ candidateActions,
1675
+ ...options.includeObservedAction ? { observedAction: point.chosenAction } : {},
1676
+ evidence: point.evidence.map((ref) => ({
1677
+ id: ref.id,
1678
+ source: ref.source,
1679
+ ...options.includeEvidenceDetail && ref.detail ? { detail: ref.detail } : {},
1680
+ ...ref.quality ? { quality: ref.quality } : {}
1681
+ })),
1682
+ context: trimText(await options.contextOf?.(point), options.maxContextChars),
1683
+ metadata: await options.metadataOf?.(point)
1684
+ });
1685
+ } catch (error) {
1686
+ diagnostics.push({
1687
+ decisionId: point.id,
1688
+ severity: "error",
1689
+ reason: `probe threw: ${errorMessage(error)}`
1690
+ });
1691
+ return { diagnostics };
1692
+ }
1693
+ const normalized = normalizeProbeResponse(response, {
1694
+ point,
1695
+ candidateActions,
1696
+ allowOutOfSetActions: options.allowOutOfSetActions ?? false
1697
+ });
1698
+ if (!normalized.record) {
1699
+ diagnostics.push(...normalized.diagnostics);
1700
+ return { diagnostics };
1701
+ }
1702
+ return {
1703
+ record: {
1704
+ probeId: options.probeId,
1705
+ decisionId: point.id,
1706
+ runId: point.runId,
1707
+ scenarioId: point.scenarioId,
1708
+ stepIndex: point.stepIndex,
1709
+ decisionKind: point.kind,
1710
+ candidateActions,
1711
+ observedAction: point.chosenAction,
1712
+ agreesWithObservedAction: normalized.record.predictedAction === point.chosenAction,
1713
+ ...options.includeOutcomeInRecord === false ? {} : { outcome: point.outcome },
1714
+ ...normalized.record
1715
+ },
1716
+ diagnostics
1717
+ };
1789
1718
  }
1790
1719
  function normalizeProbeResponse(response, options) {
1791
- const diagnostics = [];
1792
- const predictedAction = stringOrNull(response.predictedAction);
1793
- if (!predictedAction) {
1794
- diagnostics.push({
1795
- decisionId: options.point.id,
1796
- severity: "error",
1797
- reason: "missing predictedAction"
1798
- });
1799
- } else if (!options.allowOutOfSetActions && options.candidateActions.length > 0 && !options.candidateActions.includes(predictedAction)) {
1800
- diagnostics.push({
1801
- decisionId: options.point.id,
1802
- severity: "error",
1803
- reason: `predictedAction ${predictedAction} is not in candidateActions`
1804
- });
1805
- }
1806
- if (!isUnitProbability(response.confidence)) {
1807
- diagnostics.push({
1808
- decisionId: options.point.id,
1809
- severity: "error",
1810
- reason: `invalid confidence ${String(response.confidence)}`
1811
- });
1812
- }
1813
- if (response.targetProb !== void 0 && !isUnitProbability(response.targetProb)) {
1814
- diagnostics.push({
1815
- decisionId: options.point.id,
1816
- severity: "error",
1817
- reason: `invalid targetProb ${String(response.targetProb)}`
1818
- });
1819
- }
1820
- if (response.qHat !== void 0 && response.qHat !== null && !isUnitProbability(response.qHat)) {
1821
- diagnostics.push({
1822
- decisionId: options.point.id,
1823
- severity: "error",
1824
- reason: `invalid qHat ${String(response.qHat)}`
1825
- });
1826
- }
1827
- if (diagnostics.length > 0 || !predictedAction) return { diagnostics };
1828
- return {
1829
- record: {
1830
- predictedAction,
1831
- confidence: response.confidence,
1832
- ...response.beliefSummary ? { beliefSummary: trimText2(response.beliefSummary, 2e3) } : {},
1833
- uncertainty: compactStrings(response.uncertainty),
1834
- evidenceRefs: compactStrings(response.evidenceRefs),
1835
- wouldChangeMindIf: compactStrings(response.wouldChangeMindIf),
1836
- ...response.targetProb !== void 0 ? { targetProb: response.targetProb } : {},
1837
- ...response.qHat !== void 0 ? { qHat: response.qHat } : {},
1838
- ...response.metadata ? { metadata: response.metadata } : {}
1839
- },
1840
- diagnostics
1841
- };
1720
+ const diagnostics = [];
1721
+ const predictedAction = stringOrNull(response.predictedAction);
1722
+ if (!predictedAction) diagnostics.push({
1723
+ decisionId: options.point.id,
1724
+ severity: "error",
1725
+ reason: "missing predictedAction"
1726
+ });
1727
+ else if (!options.allowOutOfSetActions && options.candidateActions.length > 0 && !options.candidateActions.includes(predictedAction)) diagnostics.push({
1728
+ decisionId: options.point.id,
1729
+ severity: "error",
1730
+ reason: `predictedAction ${predictedAction} is not in candidateActions`
1731
+ });
1732
+ if (!isUnitProbability(response.confidence)) diagnostics.push({
1733
+ decisionId: options.point.id,
1734
+ severity: "error",
1735
+ reason: `invalid confidence ${String(response.confidence)}`
1736
+ });
1737
+ if (response.targetProb !== void 0 && !isUnitProbability(response.targetProb)) diagnostics.push({
1738
+ decisionId: options.point.id,
1739
+ severity: "error",
1740
+ reason: `invalid targetProb ${String(response.targetProb)}`
1741
+ });
1742
+ const hasQHatChosen = response.qHatChosen !== void 0 && response.qHatChosen !== null;
1743
+ const hasVHatTarget = response.vHatTarget !== void 0 && response.vHatTarget !== null;
1744
+ if (hasQHatChosen !== hasVHatTarget) diagnostics.push({
1745
+ decisionId: options.point.id,
1746
+ severity: "error",
1747
+ reason: "qHatChosen and vHatTarget must be supplied together"
1748
+ });
1749
+ if (hasQHatChosen && !isUnitProbability(response.qHatChosen)) diagnostics.push({
1750
+ decisionId: options.point.id,
1751
+ severity: "error",
1752
+ reason: `invalid qHatChosen ${String(response.qHatChosen)}`
1753
+ });
1754
+ if (hasVHatTarget && !isUnitProbability(response.vHatTarget)) diagnostics.push({
1755
+ decisionId: options.point.id,
1756
+ severity: "error",
1757
+ reason: `invalid vHatTarget ${String(response.vHatTarget)}`
1758
+ });
1759
+ if (diagnostics.length > 0 || !predictedAction) return { diagnostics };
1760
+ return {
1761
+ record: {
1762
+ predictedAction,
1763
+ confidence: response.confidence,
1764
+ ...response.beliefSummary ? { beliefSummary: trimText(response.beliefSummary, 2e3) } : {},
1765
+ uncertainty: compactStrings(response.uncertainty),
1766
+ evidenceRefs: compactStrings(response.evidenceRefs),
1767
+ wouldChangeMindIf: compactStrings(response.wouldChangeMindIf),
1768
+ ...response.targetProb !== void 0 ? { targetProb: response.targetProb } : {},
1769
+ ...response.qHatChosen !== void 0 ? { qHatChosen: response.qHatChosen } : {},
1770
+ ...response.vHatTarget !== void 0 ? { vHatTarget: response.vHatTarget } : {},
1771
+ ...response.metadata ? { metadata: response.metadata } : {}
1772
+ },
1773
+ diagnostics
1774
+ };
1842
1775
  }
1843
1776
  function summarizeShadowProbe(attempted, records) {
1844
- const confidences = records.map((record) => record.confidence);
1845
- const agreements = records.filter((record) => record.agreesWithObservedAction).length;
1846
- return {
1847
- attempted,
1848
- completed: records.length,
1849
- dropped: attempted - records.length,
1850
- withOutcome: records.filter((record) => record.outcome !== void 0).length,
1851
- withTargetProb: records.filter((record) => record.targetProb !== void 0).length,
1852
- meanConfidence: confidences.length > 0 ? mean3(confidences) : null,
1853
- observedAgreementRate: records.length > 0 ? agreements / records.length : null
1854
- };
1777
+ const confidences = records.map((record) => record.confidence);
1778
+ const agreements = records.filter((record) => record.agreesWithObservedAction).length;
1779
+ return {
1780
+ attempted,
1781
+ completed: records.length,
1782
+ dropped: attempted - records.length,
1783
+ withOutcome: records.filter((record) => record.outcome !== void 0).length,
1784
+ withTargetProb: records.filter((record) => record.targetProb !== void 0).length,
1785
+ meanConfidence: confidences.length > 0 ? mean(confidences) : null,
1786
+ observedAgreementRate: records.length > 0 ? agreements / records.length : null
1787
+ };
1855
1788
  }
1856
1789
  function isUnitProbability(value) {
1857
- return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
1790
+ return typeof value === "number" && Number.isFinite(value) && value >= 0 && value <= 1;
1858
1791
  }
1859
1792
  function boundedInteger(value, min, max) {
1860
- if (!Number.isFinite(value)) return min;
1861
- return Math.max(min, Math.min(max, Math.floor(value)));
1793
+ if (!Number.isFinite(value)) return min;
1794
+ return Math.max(min, Math.min(max, Math.floor(value)));
1862
1795
  }
1863
1796
  function compactStrings(values, maxItems = 12) {
1864
- if (!Array.isArray(values)) return [];
1865
- return values.filter((value) => typeof value === "string" && value.length > 0).slice(0, maxItems).map((value) => trimText2(value, 500) ?? "").filter(Boolean);
1797
+ if (!Array.isArray(values)) return [];
1798
+ return values.filter((value) => typeof value === "string" && value.length > 0).slice(0, maxItems).map((value) => trimText(value, 500) ?? "").filter(Boolean);
1866
1799
  }
1867
- function uniqueStrings2(values) {
1868
- return [...new Set(values.filter((value) => value.length > 0))];
1800
+ function uniqueStrings(values) {
1801
+ return [...new Set(values.filter((value) => value.length > 0))];
1869
1802
  }
1870
1803
  function stringOrNull(value) {
1871
- return typeof value === "string" && value.length > 0 ? value : null;
1872
- }
1873
- function trimText2(value, maxChars = DEFAULT_MAX_CONTEXT_CHARS2) {
1874
- if (!value) return void 0;
1875
- return value.length > maxChars ? value.slice(value.length - maxChars) : value;
1876
- }
1877
- function mean3(values) {
1878
- return values.reduce((sum, value) => sum + value, 0) / values.length;
1879
- }
1880
- function errorMessage2(error) {
1881
- return error instanceof Error ? error.message : String(error);
1882
- }
1883
- export {
1884
- BELIEF_DECISION_KINDS,
1885
- BELIEF_EVALUATION_CRITERIA,
1886
- BELIEF_EVIDENCE_QUALITIES,
1887
- BELIEF_EVIDENCE_SOURCES,
1888
- analyzeBeliefDecisionCorpus,
1889
- analyzeBeliefPolicy,
1890
- beliefDecisionsToOffPolicyTrajectories,
1891
- buildBeliefDecisionResearchEvidencePacket,
1892
- buildCodeAgentBeliefEvidenceCorpus,
1893
- buildRuntimeBeliefPhase0Measurement,
1894
- buildRuntimeBenchmarkBeliefPhase0Measurement,
1895
- calibrateBeliefDecisions,
1896
- createBeliefRuntimeHookCollector,
1897
- embeddedBeliefOpeTargetPolicy,
1898
- evaluateBeliefOffPolicy,
1899
- evaluateBeliefSelectivePolicy,
1900
- extractBeliefDecisionPoints,
1901
- extractCodeAgentBeliefDecisionPoints,
1902
- formatBeliefShadowProbePrompt,
1903
- inventoryBeliefDecisionPoints,
1904
- isBeliefDecisionKind,
1905
- isBeliefEvidenceSource,
1906
- runBeliefShadowProbe,
1907
- runtimeDecisionPointToBeliefDecisionPoint,
1908
- runtimeDecisionPointToBeliefShadowProbeInput,
1909
- selectBeliefDecisionTarget,
1910
- thresholdSelectivePolicy
1911
- };
1804
+ return typeof value === "string" && value.length > 0 ? value : null;
1805
+ }
1806
+ function trimText(value, maxChars = DEFAULT_MAX_CONTEXT_CHARS) {
1807
+ if (!value) return void 0;
1808
+ return value.length > maxChars ? value.slice(value.length - maxChars) : value;
1809
+ }
1810
+ function mean(values) {
1811
+ return values.reduce((sum, value) => sum + value, 0) / values.length;
1812
+ }
1813
+ function errorMessage(error) {
1814
+ return error instanceof Error ? error.message : String(error);
1815
+ }
1816
+ //#endregion
1817
+ export { BELIEF_DECISION_KINDS, BELIEF_EVALUATION_CRITERIA, BELIEF_EVIDENCE_QUALITIES, BELIEF_EVIDENCE_SOURCES, analyzeBeliefDecisionCorpus, analyzeBeliefPolicy, beliefDecisionsToOffPolicyTrajectories, buildBeliefDecisionResearchEvidencePacket, buildCodeAgentBeliefEvidenceCorpus, buildRuntimeBeliefPhase0Measurement, buildRuntimeBenchmarkBeliefPhase0Measurement, calibrateBeliefDecisions, createBeliefRuntimeHookCollector, embeddedBeliefOpeTargetPolicy, evaluateBeliefOffPolicy, evaluateBeliefSelectivePolicy, extractBeliefDecisionPoints, extractCodeAgentBeliefDecisionPoints, formatBeliefShadowProbePrompt, inventoryBeliefDecisionPoints, isBeliefDecisionKind, isBeliefEvidenceSource, runBeliefShadowProbe, runtimeDecisionPointToBeliefDecisionPoint, runtimeDecisionPointToBeliefShadowProbeInput, selectBeliefDecisionTarget, thresholdSelectivePolicy };
1818
+
1912
1819
  //# sourceMappingURL=index.js.map