@tangle-network/agent-eval 0.129.0 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (427) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +1 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/package.json +17 -9
  301. package/dist/benchmarks/index.js.map +0 -1
  302. package/dist/campaign/index.js.map +0 -1
  303. package/dist/chunk-2QU3YOPR.js +0 -7374
  304. package/dist/chunk-2QU3YOPR.js.map +0 -1
  305. package/dist/chunk-3OCR4R5I.js +0 -728
  306. package/dist/chunk-3OCR4R5I.js.map +0 -1
  307. package/dist/chunk-3RF76KTD.js +0 -84
  308. package/dist/chunk-3RF76KTD.js.map +0 -1
  309. package/dist/chunk-56TAVBOK.js +0 -698
  310. package/dist/chunk-5DTSBUL2.js +0 -159
  311. package/dist/chunk-5DTSBUL2.js.map +0 -1
  312. package/dist/chunk-7FO3TNPI.js +0 -232
  313. package/dist/chunk-7FO3TNPI.js.map +0 -1
  314. package/dist/chunk-7ZZMD7UK.js +0 -386
  315. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  316. package/dist/chunk-BOD4O7OF.js +0 -40
  317. package/dist/chunk-BOD4O7OF.js.map +0 -1
  318. package/dist/chunk-BSO5JDQH.js +0 -2335
  319. package/dist/chunk-BSO5JDQH.js.map +0 -1
  320. package/dist/chunk-C6LXANRU.js +0 -1550
  321. package/dist/chunk-C6LXANRU.js.map +0 -1
  322. package/dist/chunk-DODXQREJ.js +0 -752
  323. package/dist/chunk-DODXQREJ.js.map +0 -1
  324. package/dist/chunk-DRYIUNWY.js +0 -622
  325. package/dist/chunk-DRYIUNWY.js.map +0 -1
  326. package/dist/chunk-E7QXT7SX.js +0 -183
  327. package/dist/chunk-E7QXT7SX.js.map +0 -1
  328. package/dist/chunk-EG66UGL4.js +0 -341
  329. package/dist/chunk-EG66UGL4.js.map +0 -1
  330. package/dist/chunk-FXTVJPYD.js +0 -576
  331. package/dist/chunk-FXTVJPYD.js.map +0 -1
  332. package/dist/chunk-G7MGMCZD.js +0 -153
  333. package/dist/chunk-G7MGMCZD.js.map +0 -1
  334. package/dist/chunk-GGE4NNQT.js +0 -65
  335. package/dist/chunk-GGE4NNQT.js.map +0 -1
  336. package/dist/chunk-H23X7XKK.js +0 -181
  337. package/dist/chunk-H23X7XKK.js.map +0 -1
  338. package/dist/chunk-HHWE3POT.js +0 -94
  339. package/dist/chunk-HHWE3POT.js.map +0 -1
  340. package/dist/chunk-HPWUNB47.js +0 -289
  341. package/dist/chunk-HPWUNB47.js.map +0 -1
  342. package/dist/chunk-IYCLP2N2.js +0 -766
  343. package/dist/chunk-IYCLP2N2.js.map +0 -1
  344. package/dist/chunk-JHCHEVET.js +0 -274
  345. package/dist/chunk-JHCHEVET.js.map +0 -1
  346. package/dist/chunk-JQSF5DQT.js +0 -701
  347. package/dist/chunk-JQSF5DQT.js.map +0 -1
  348. package/dist/chunk-K4DBDHLK.js +0 -158
  349. package/dist/chunk-K4DBDHLK.js.map +0 -1
  350. package/dist/chunk-K6N6XJJX.js +0 -306
  351. package/dist/chunk-K6N6XJJX.js.map +0 -1
  352. package/dist/chunk-M4YBQKIJ.js +0 -1040
  353. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  354. package/dist/chunk-MA6HLL3S.js +0 -65
  355. package/dist/chunk-MA6HLL3S.js.map +0 -1
  356. package/dist/chunk-MAZ26DC7.js +0 -99
  357. package/dist/chunk-MAZ26DC7.js.map +0 -1
  358. package/dist/chunk-NPCTHQIO.js +0 -91
  359. package/dist/chunk-NPCTHQIO.js.map +0 -1
  360. package/dist/chunk-NY44NC4A.js +0 -1056
  361. package/dist/chunk-NY44NC4A.js.map +0 -1
  362. package/dist/chunk-OIUOT4QD.js +0 -44
  363. package/dist/chunk-OIUOT4QD.js.map +0 -1
  364. package/dist/chunk-ONWEPEDO.js +0 -57
  365. package/dist/chunk-ONWEPEDO.js.map +0 -1
  366. package/dist/chunk-OWN5NPMC.js +0 -152
  367. package/dist/chunk-OWN5NPMC.js.map +0 -1
  368. package/dist/chunk-P6FYH6K4.js +0 -1161
  369. package/dist/chunk-P6FYH6K4.js.map +0 -1
  370. package/dist/chunk-PC4UYEBM.js +0 -166
  371. package/dist/chunk-PC4UYEBM.js.map +0 -1
  372. package/dist/chunk-PC5DOSM7.js +0 -579
  373. package/dist/chunk-PC5DOSM7.js.map +0 -1
  374. package/dist/chunk-PXE2VKMX.js +0 -140
  375. package/dist/chunk-PXE2VKMX.js.map +0 -1
  376. package/dist/chunk-PZ5AY32C.js +0 -10
  377. package/dist/chunk-PZ5AY32C.js.map +0 -1
  378. package/dist/chunk-QB6BDBP2.js +0 -4464
  379. package/dist/chunk-QB6BDBP2.js.map +0 -1
  380. package/dist/chunk-RXHCETDZ.js +0 -536
  381. package/dist/chunk-RXHCETDZ.js.map +0 -1
  382. package/dist/chunk-RZTMDUO7.js +0 -49
  383. package/dist/chunk-RZTMDUO7.js.map +0 -1
  384. package/dist/chunk-SFLLL76A.js +0 -669
  385. package/dist/chunk-SFLLL76A.js.map +0 -1
  386. package/dist/chunk-SZLVEKMJ.js +0 -1446
  387. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  388. package/dist/chunk-T4SQEITX.js +0 -95
  389. package/dist/chunk-T4SQEITX.js.map +0 -1
  390. package/dist/chunk-T6RLYGAD.js +0 -158
  391. package/dist/chunk-T6RLYGAD.js.map +0 -1
  392. package/dist/chunk-TJVT4QFF.js +0 -911
  393. package/dist/chunk-TJVT4QFF.js.map +0 -1
  394. package/dist/chunk-TQ7LNKZ3.js +0 -136
  395. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  396. package/dist/chunk-U4L7JRPZ.js +0 -1706
  397. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  398. package/dist/chunk-U4PHLT2N.js +0 -419
  399. package/dist/chunk-U4PHLT2N.js.map +0 -1
  400. package/dist/chunk-VCZ5FQYW.js +0 -928
  401. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  402. package/dist/chunk-VI2UW6B6.js +0 -162
  403. package/dist/chunk-VI2UW6B6.js.map +0 -1
  404. package/dist/chunk-VQMK5FMP.js +0 -247
  405. package/dist/chunk-VQMK5FMP.js.map +0 -1
  406. package/dist/chunk-WGXIEX7P.js +0 -116
  407. package/dist/chunk-WGXIEX7P.js.map +0 -1
  408. package/dist/chunk-WVATSFCP.js +0 -1553
  409. package/dist/chunk-WVATSFCP.js.map +0 -1
  410. package/dist/chunk-X4YIBDER.js +0 -1662
  411. package/dist/chunk-X4YIBDER.js.map +0 -1
  412. package/dist/chunk-YQN4ICPP.js +0 -355
  413. package/dist/chunk-YQN4ICPP.js.map +0 -1
  414. package/dist/chunk-ZET2UAYW.js +0 -89
  415. package/dist/chunk-ZET2UAYW.js.map +0 -1
  416. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  417. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  418. package/dist/control.js.map +0 -1
  419. package/dist/hosted/index.js.map +0 -1
  420. package/dist/matrix/index.js.map +0 -1
  421. package/dist/reporting.js.map +0 -1
  422. package/dist/rollout/index.js.map +0 -1
  423. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  424. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  425. package/dist/supervisor-run/index.js.map +0 -1
  426. package/dist/traces.js.map +0 -1
  427. package/dist/wire/index.js.map +0 -1
@@ -0,0 +1,2579 @@
1
+ import { i as CostLedger } from "./cost-ledger-DIgQUFZZ.js";
2
+ import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, n as LlmClient, s as callLlm, u as costReceiptFromLlmError } from "./llm-client--GR4JbZE.js";
3
+ import { LLM_CONTEXT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_OUTPUT_TOKEN_ATTR_KEYS, TOOL_NAME_ATTR_KEYS } from "./trace-attributes.js";
4
+ import { t as executionTrackByLane } from "./execution-tracks-CpgFPpS5.js";
5
+ import { _ as spanEpochMillis, r as runTraceAnalysisLoop, t as buildTraceAnalystTools } from "./tools-BmuN627J.js";
6
+ import { i as combineAbortSignals } from "./run-score-iEEAWiBY.js";
7
+ import { ai } from "@ax-llm/ax";
8
+ import { z } from "zod";
9
+ import { createHash, randomUUID } from "node:crypto";
10
+ //#region src/analyst/ax-service.ts
11
+ const configuredModels = /* @__PURE__ */ new WeakMap();
12
+ /**
13
+ * Construct the `AxAIService` an analyst kind calls through
14
+ * (`createTraceAnalystKind({ ai })`).
15
+ *
16
+ * Ax's `ai()` pins `config.model` to the OpenAI catalog enum, but every
17
+ * OpenAI-compatible router an analyst points at (router.tangle.tools,
18
+ * cli-bridge) accepts arbitrary model ids (claude-code/sonnet, openai/gpt-5.4,
19
+ * …). Consumers were each re-rolling `ai({ name, apiKey, apiURL, config })`
20
+ * behind an `as (a: any) => any` cast to dodge the enum; this is the one
21
+ * canonical constructor so they don't have to — and don't take a direct
22
+ * `@ax-llm/ax` dependency for it.
23
+ */
24
+ function createAnalystAi(config) {
25
+ const model = config.model.trim();
26
+ if (!model) throw new TypeError("createAnalystAi: model must be a non-empty string");
27
+ const service = ai({
28
+ name: config.provider ?? "openai",
29
+ apiKey: config.apiKey,
30
+ ...config.baseUrl ? { apiURL: config.baseUrl } : {},
31
+ ...config.headers ? { headers: config.headers } : {},
32
+ config: { model }
33
+ });
34
+ configuredModels.set(service, model);
35
+ return service;
36
+ }
37
+ function getConfiguredAnalystModel(service) {
38
+ return configuredModels.get(service);
39
+ }
40
+ /** Resolve the model before paid work so every request can be bounded and attributed. */
41
+ function resolveAnalystModel(service, override) {
42
+ if (override !== void 0) {
43
+ const model = override.trim();
44
+ if (!model) throw new TypeError("createTraceAnalystKind: model must be a non-empty string");
45
+ return model;
46
+ }
47
+ const model = getConfiguredAnalystModel(service)?.trim();
48
+ if (!model) throw new TypeError("createTraceAnalystKind: model is required for Ax services not created by createAnalystAi()");
49
+ return model;
50
+ }
51
+ //#endregion
52
+ //#region src/analyst/chat-client.ts
53
+ /**
54
+ * Provider-neutral chat contract for every model call made by agent-eval.
55
+ *
56
+ * Callers choose the transport at the package boundary with `createChatClient`.
57
+ * Evaluation code receives canonical requests and results without importing a
58
+ * provider SDK.
59
+ */
60
+ /**
61
+ * Build a ChatClient bound to a specific transport. The returned client
62
+ * is safe to share across analysts in a single registry run.
63
+ */
64
+ function createChatClient(opts) {
65
+ switch (opts.transport) {
66
+ case "router": return wrapLlmClient(opts.transport, opts.defaultModel, new LlmClient({
67
+ baseUrl: opts.baseUrl ?? "https://router.tangle.tools/v1",
68
+ apiKey: opts.apiKey,
69
+ maximumAttempts: opts.maximumAttempts
70
+ }));
71
+ case "cli-bridge": return wrapLlmClient(opts.transport, opts.defaultModel, new LlmClient({
72
+ baseUrl: opts.baseUrl ?? "http://127.0.0.1:3344/v1",
73
+ apiKey: opts.bearer ?? "",
74
+ maximumAttempts: opts.maximumAttempts
75
+ }));
76
+ case "direct-provider": return wrapLlmClient(opts.transport, opts.defaultModel, new LlmClient({
77
+ baseUrl: opts.baseUrl,
78
+ apiKey: opts.apiKey,
79
+ maximumAttempts: opts.maximumAttempts
80
+ }));
81
+ case "sandbox-sdk": return {
82
+ transport: "sandbox-sdk",
83
+ defaultModel: opts.defaultModel,
84
+ maximumAttempts: opts.maximumAttempts,
85
+ chat: async (req, callOpts) => opts.chat(resolveModel(req, opts.defaultModel), callOpts)
86
+ };
87
+ case "custom": return {
88
+ transport: "custom",
89
+ defaultModel: opts.defaultModel,
90
+ maximumAttempts: opts.maximumAttempts,
91
+ chat: async (req, callOpts) => opts.chat(resolveModel(req, opts.defaultModel), callOpts)
92
+ };
93
+ case "mock": return {
94
+ transport: "mock",
95
+ defaultModel: opts.defaultModel,
96
+ maximumAttempts: 1,
97
+ chat: async (req, callOpts) => opts.handler(resolveModel(req, opts.defaultModel), callOpts)
98
+ };
99
+ }
100
+ }
101
+ function wrapLlmClient(transport, defaultModel, inner) {
102
+ return {
103
+ transport,
104
+ defaultModel,
105
+ maximumAttempts: inner.maximumAttempts,
106
+ chat: (req, callOpts) => {
107
+ const request = {
108
+ model: resolveModel(req, defaultModel).model,
109
+ messages: req.messages,
110
+ jsonMode: req.jsonMode,
111
+ jsonSchema: req.jsonSchema,
112
+ temperature: req.temperature,
113
+ maxTokens: req.maxTokens,
114
+ thinking: req.thinking,
115
+ timeoutMs: req.timeoutMs
116
+ };
117
+ return inner.call(request, {
118
+ signal: callOpts?.signal,
119
+ idempotencyKey: callOpts?.idempotencyKey
120
+ });
121
+ }
122
+ };
123
+ }
124
+ function resolveModel(req, defaultModel) {
125
+ if (req.model) return req;
126
+ if (!defaultModel) throw new Error("ChatClient.chat: no model on request and no defaultModel on the client. Either pass req.model or bind defaultModel at createChatClient().");
127
+ return {
128
+ ...req,
129
+ model: defaultModel
130
+ };
131
+ }
132
+ //#endregion
133
+ //#region src/trace-analyst/behavioral-metrics.ts
134
+ /**
135
+ * Deterministic behavioral metrics over OTLP spans — pure arithmetic, no LLM.
136
+ *
137
+ * These are the model-independent multiplier: the four trace-quality signals a
138
+ * tolerant analyzer (e.g. HALO) re-derives per run inside the model — token
139
+ * growth, output decay, tool monoculture, missing self-verification — computed
140
+ * here once, in TypeScript, with zero model judgment. A finding that falls out
141
+ * of arithmetic is trivially model-agnostic and cannot hallucinate the trend.
142
+ *
143
+ * General, not trace-specific: the detectors key off token trajectories and
144
+ * tool usage present in any agentic OTLP trace, not any one benchmark.
145
+ */
146
+ /** ≥ this input-token growth ratio across a run, with no compression, fires. */
147
+ const INPUT_GROWTH_FACTOR = 3;
148
+ /** Tool-usage signals need at least this many calls to be meaningful. */
149
+ const MIN_TOOL_CALLS = 3;
150
+ /** Tool names that read or check state count as self-verification, not mutation.
151
+ * Covers the inspect verbs plus the read/search tools real harnesses use to
152
+ * verify (Claude Code Read/Grep/Glob, codex read_file/ls/cat, git status/diff,
153
+ * test/lint). A pure shell tool (Bash/exec_command) is intentionally NOT matched
154
+ * — its name can't tell a `pytest` from an `rm`. */
155
+ const VERIFY_RE = /verif|eval|inspect|check|assert|validat|review|confirm|read|grep|glob|search|view|\blist\b|\bls\b|\bcat\b|\bfind\b|diff|status|\btest|lint|typecheck/i;
156
+ function num(v) {
157
+ return typeof v === "number" && Number.isFinite(v) ? v : null;
158
+ }
159
+ function numAttr(attrs, keys) {
160
+ for (const key of keys) {
161
+ const value = num(attrs[key]);
162
+ if (value !== null) return value;
163
+ }
164
+ return null;
165
+ }
166
+ function inputTokensOf(s) {
167
+ const exactContext = num(s.attributes[LLM_CONTEXT_TOKENS]);
168
+ if (exactContext !== null) return exactContext;
169
+ return numAttr(s.attributes, LLM_INPUT_TOKEN_ATTR_KEYS) ?? num(s.attributes["llm.usage.input_tokens"]);
170
+ }
171
+ function outputTokensOf(s) {
172
+ return numAttr(s.attributes, LLM_OUTPUT_TOKEN_ATTR_KEYS) ?? num(s.attributes["llm.usage.output_tokens"]);
173
+ }
174
+ function stepOf(s) {
175
+ return num(s.attributes.step);
176
+ }
177
+ function toolNameOf(s) {
178
+ if (s.tool_name) return s.tool_name;
179
+ for (const key of TOOL_NAME_ATTR_KEYS) {
180
+ const t = s.attributes[key];
181
+ if (typeof t === "string" && t.length > 0) return t;
182
+ }
183
+ return null;
184
+ }
185
+ /**
186
+ * Reduce a span list to behavioral metrics + fired suboptimality signals.
187
+ * Pure + deterministic: same spans → same output, on any machine, no model.
188
+ */
189
+ function computeTraceMetrics(spans) {
190
+ const traceIds = new Set(spans.map((span) => span.trace_id));
191
+ if (traceIds.size > 1) throw new Error(`computeTraceMetrics: expected spans from one trace, received ${traceIds.size} traces`);
192
+ const traceId = traceIds.values().next().value ?? null;
193
+ const samples = spans.map((span) => ({
194
+ span,
195
+ input: inputTokensOf(span),
196
+ output: outputTokensOf(span),
197
+ step: stepOf(span)
198
+ }));
199
+ const llmSamples = samples.filter((sample) => sample.span.kind === "LLM");
200
+ const tokenSamples = llmSamples.length > 0 ? llmSamples : samples.filter((sample) => sample.input !== null || sample.output !== null);
201
+ const tokenSequences = buildTokenSequences(tokenSamples, spans);
202
+ const primarySequence = tokenSequences[0];
203
+ const inputTokenTrajectory = primarySequence?.inputTokenTrajectory.filter((value) => value !== null) ?? [];
204
+ const outputTokenTrajectory = primarySequence?.outputTokenTrajectory.filter((value) => value !== null) ?? [];
205
+ const toolHistogram = {};
206
+ let hasSelfVerification = false;
207
+ for (const s of spans) {
208
+ const tool = toolNameOf(s);
209
+ if (tool) {
210
+ toolHistogram[tool] = (toolHistogram[tool] ?? 0) + 1;
211
+ if (VERIFY_RE.test(tool)) hasSelfVerification = true;
212
+ }
213
+ }
214
+ const totalToolCalls = Object.values(toolHistogram).reduce((a, b) => a + b, 0);
215
+ const distinctTools = Object.keys(toolHistogram).length;
216
+ const toolDiversityRatio = totalToolCalls === 0 ? 1 : distinctTools / totalToolCalls;
217
+ const signals = [];
218
+ const seenTokenSignals = /* @__PURE__ */ new Set();
219
+ for (const sequence of tokenSequences) for (const signal of tokenSignals(sequence)) {
220
+ if (seenTokenSignals.has(signal.code)) continue;
221
+ seenTokenSignals.add(signal.code);
222
+ signals.push(signal);
223
+ }
224
+ if (totalToolCalls >= MIN_TOOL_CALLS && distinctTools === 1) {
225
+ const only = Object.keys(toolHistogram)[0];
226
+ signals.push({
227
+ code: "single-tool-dependency",
228
+ severity: "medium",
229
+ detail: `All ${totalToolCalls} observed tool calls are \`${only}\`; no alternate tool call was observed.`,
230
+ evidence: {
231
+ tool: only,
232
+ calls: totalToolCalls,
233
+ distinct_tools: 1
234
+ }
235
+ });
236
+ }
237
+ if (totalToolCalls >= MIN_TOOL_CALLS && !hasSelfVerification) signals.push({
238
+ code: "no-self-verification",
239
+ severity: "medium",
240
+ detail: `${totalToolCalls} tool calls were observed without a verification-named tool call.`,
241
+ evidence: {
242
+ tool_calls: totalToolCalls,
243
+ verification_calls: 0
244
+ }
245
+ });
246
+ return {
247
+ traceId,
248
+ llmCallCount: tokenSamples.length,
249
+ tokenSequences,
250
+ inputTokenTrajectory,
251
+ outputTokenTrajectory,
252
+ toolHistogram,
253
+ totalToolCalls,
254
+ distinctTools,
255
+ toolDiversityRatio,
256
+ hasSelfVerification,
257
+ signals
258
+ };
259
+ }
260
+ function buildTokenSequences(samples, spans) {
261
+ const executionScopeFor = createTokenExecutionScopeResolver(new Map(spans.map((span) => [span.span_id, span])));
262
+ const scopedSamples = samples.map((sample) => ({
263
+ sample,
264
+ ...executionScopeFor(sample.span)
265
+ }));
266
+ const trackByLane = executionTrackByLane(scopedSamples);
267
+ const byTrack = /* @__PURE__ */ new Map();
268
+ for (const scoped of scopedSamples) {
269
+ const trackId = trackByLane.get(scoped.key);
270
+ const track = byTrack.get(trackId) ?? {
271
+ scopeId: scoped.scopeId,
272
+ samples: []
273
+ };
274
+ track.samples.push(scoped.sample);
275
+ byTrack.set(trackId, track);
276
+ }
277
+ const sequences = [];
278
+ for (const { scopeId, samples: tracked } of byTrack.values()) {
279
+ const runs = serialTokenRuns([...tracked].sort(compareTokenSamples));
280
+ runs.forEach((run, index) => {
281
+ sequences.push({
282
+ scopeId: runs.length === 1 ? scopeId : `${scopeId}#${index + 1}`,
283
+ spanIds: run.map((sample) => sample.span.span_id),
284
+ inputTokenTrajectory: run.map((sample) => sample.input),
285
+ outputTokenTrajectory: run.map((sample) => sample.output)
286
+ });
287
+ });
288
+ }
289
+ return sequences.sort((a, b) => b.spanIds.length - a.spanIds.length || a.scopeId.localeCompare(b.scopeId) || a.spanIds[0].localeCompare(b.spanIds[0]));
290
+ }
291
+ function createTokenExecutionScopeResolver(spansById) {
292
+ const cache = /* @__PURE__ */ new Map();
293
+ return (span) => {
294
+ const ancestry = resolveAncestorScope(span.parent_span_id, spansById, cache);
295
+ const rootId = ancestry.rootId;
296
+ const scopeId = ancestry.agentId ? `span:${ancestry.agentId}` : ancestry.missingParentId ? `parent:${ancestry.missingParentId}` : rootId ? `root:${rootId}` : span.agent_name ? `agent:${span.agent_name}` : `trace:${span.trace_id}`;
297
+ const scopeSpanId = ancestry.agentId ?? ancestry.missingParentId ?? rootId;
298
+ const laneSpan = ancestry.laneSpanId ? spansById.get(ancestry.laneSpanId) : void 0;
299
+ const direct = scopeSpanId === null || ancestry.laneSpanId === scopeSpanId;
300
+ const timedSpan = direct ? span : laneSpan;
301
+ return {
302
+ key: JSON.stringify([scopeId, direct ? span.span_id : ancestry.laneSpanId]),
303
+ scopeKey: scopeId,
304
+ scopeId,
305
+ start: timedSpan ? spanEpochMillis(timedSpan.start_time) : null,
306
+ end: timedSpan ? spanEpochMillis(timedSpan.end_time) : null
307
+ };
308
+ };
309
+ }
310
+ function resolveAncestorScope(startId, spansById, cache) {
311
+ const empty = {
312
+ agentId: null,
313
+ rootId: null,
314
+ missingParentId: null,
315
+ laneSpanId: null
316
+ };
317
+ if (!startId) return empty;
318
+ const path = [];
319
+ const pathIndex = /* @__PURE__ */ new Map();
320
+ let currentId = startId;
321
+ let resolved = empty;
322
+ while (currentId) {
323
+ const cached = cache.get(currentId);
324
+ if (cached) {
325
+ resolved = cached;
326
+ break;
327
+ }
328
+ const cycleStart = pathIndex.get(currentId);
329
+ if (cycleStart !== void 0) {
330
+ resolved = {
331
+ agentId: null,
332
+ rootId: [...path.slice(cycleStart)].sort()[0],
333
+ missingParentId: null,
334
+ laneSpanId: [...path.slice(cycleStart)].sort()[0]
335
+ };
336
+ break;
337
+ }
338
+ const current = spansById.get(currentId);
339
+ if (!current) {
340
+ resolved = {
341
+ agentId: null,
342
+ rootId: null,
343
+ missingParentId: currentId,
344
+ laneSpanId: currentId
345
+ };
346
+ break;
347
+ }
348
+ if (current.kind === "AGENT") {
349
+ resolved = {
350
+ agentId: current.span_id,
351
+ rootId: null,
352
+ missingParentId: null,
353
+ laneSpanId: current.span_id
354
+ };
355
+ break;
356
+ }
357
+ pathIndex.set(currentId, path.length);
358
+ path.push(currentId);
359
+ currentId = current.parent_span_id;
360
+ }
361
+ for (let index = path.length - 1; index >= 0; index -= 1) {
362
+ if (resolved.agentId === null && resolved.rootId === null) resolved = {
363
+ ...resolved,
364
+ rootId: path[index],
365
+ laneSpanId: path[index]
366
+ };
367
+ else if (resolved.laneSpanId === (resolved.agentId ?? resolved.missingParentId ?? resolved.rootId)) resolved = {
368
+ ...resolved,
369
+ laneSpanId: path[index]
370
+ };
371
+ cache.set(path[index], resolved);
372
+ }
373
+ return resolved;
374
+ }
375
+ function compareTokenSamples(a, b) {
376
+ const aStart = spanEpochMillis(a.span.start_time);
377
+ const bStart = spanEpochMillis(b.span.start_time);
378
+ if (aStart === null && bStart !== null) return 1;
379
+ if (aStart !== null && bStart === null) return -1;
380
+ if (aStart !== null && bStart !== null && aStart !== bStart) return aStart - bStart;
381
+ if (a.step !== null && b.step !== null && a.step !== b.step) return a.step - b.step;
382
+ return a.span.span_id.localeCompare(b.span.span_id);
383
+ }
384
+ function serialTokenRuns(ordered) {
385
+ const runs = [];
386
+ let serial = [];
387
+ let overlap = [];
388
+ let overlapEnd = Number.NEGATIVE_INFINITY;
389
+ const flushSerial = () => {
390
+ if (serial.length > 0) runs.push(serial);
391
+ serial = [];
392
+ };
393
+ const flushOverlap = () => {
394
+ if (overlap.length === 1) serial.push(overlap[0]);
395
+ else if (overlap.length > 1) {
396
+ flushSerial();
397
+ for (const sample of overlap) runs.push([sample]);
398
+ }
399
+ overlap = [];
400
+ overlapEnd = Number.NEGATIVE_INFINITY;
401
+ };
402
+ for (const sample of ordered) {
403
+ const start = spanEpochMillis(sample.span.start_time);
404
+ const end = spanEpochMillis(sample.span.end_time);
405
+ if (start === null || end === null || sample.span.duration_ms <= 0 || end < start) {
406
+ flushOverlap();
407
+ flushSerial();
408
+ runs.push([sample]);
409
+ continue;
410
+ }
411
+ if (overlap.length > 0 && start >= overlapEnd) flushOverlap();
412
+ overlap.push(sample);
413
+ overlapEnd = Math.max(overlapEnd, end);
414
+ }
415
+ flushOverlap();
416
+ flushSerial();
417
+ return runs;
418
+ }
419
+ function tokenSignals(sequence) {
420
+ const signals = [];
421
+ const inputs = sequence.inputTokenTrajectory;
422
+ const outputs = sequence.outputTokenTrajectory;
423
+ if (inputs.length >= 3 && inputs.every((value) => value !== null)) {
424
+ const first = inputs[0];
425
+ const last = inputs[inputs.length - 1];
426
+ const isMonotonic = everyAdjacent(inputs, (previous, current) => current >= previous);
427
+ const growthFromZero = first === 0 && last > 0;
428
+ const growth = growthFromZero ? Infinity : first > 0 ? last / first : 0;
429
+ if (isMonotonic && last > first && growth >= INPUT_GROWTH_FACTOR) {
430
+ const growthLabel = growthFromZero ? "0→nonzero (unbounded)" : `${growth.toFixed(1)}x`;
431
+ signals.push({
432
+ code: "monotonic-input-growth",
433
+ severity: "high",
434
+ detail: `LLM input tokens grew ${growthLabel} (${first}→${last}) across ${inputs.length} serial calls without an intervening decrease.`,
435
+ evidence: {
436
+ first,
437
+ last,
438
+ growth_x: growthFromZero ? "unbounded" : Number(growth.toFixed(2)),
439
+ calls: inputs.length,
440
+ scope: sequence.scopeId,
441
+ first_span_id: sequence.spanIds[0],
442
+ last_span_id: sequence.spanIds[sequence.spanIds.length - 1]
443
+ }
444
+ });
445
+ }
446
+ }
447
+ if (inputs.length >= 3 && inputs.length === outputs.length && inputs.every((value) => value !== null) && outputs.every((value) => value !== null)) {
448
+ const first = outputs[0];
449
+ const last = outputs[outputs.length - 1];
450
+ const inputIsMonotonic = everyAdjacent(inputs, (previous, current) => current >= previous);
451
+ const outputIsMonotonic = everyAdjacent(outputs, (previous, current) => current <= previous);
452
+ const inputGrew = inputs[inputs.length - 1] > inputs[0];
453
+ if (inputIsMonotonic && inputGrew && outputIsMonotonic && last < first) signals.push({
454
+ code: "output-length-decay",
455
+ severity: "medium",
456
+ detail: `LLM output tokens shrank ${first}→${last} over ${outputs.length} serial calls while input tokens increased monotonically.`,
457
+ evidence: {
458
+ first,
459
+ last,
460
+ calls: outputs.length,
461
+ scope: sequence.scopeId,
462
+ first_span_id: sequence.spanIds[0],
463
+ last_span_id: sequence.spanIds[sequence.spanIds.length - 1]
464
+ }
465
+ });
466
+ }
467
+ return signals;
468
+ }
469
+ function everyAdjacent(values, predicate) {
470
+ return values.slice(1).every((current, index) => predicate(values[index], current));
471
+ }
472
+ //#endregion
473
+ //#region src/analyst/types.ts
474
+ /**
475
+ * Analyst contract — the missing orchestration layer over agent-eval's
476
+ * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
477
+ * SemanticConceptJudge, JudgeFn, ...).
478
+ *
479
+ * Each existing primitive returns its own output shape. The Analyst
480
+ * contract is the single envelope every primitive lifts into, so a
481
+ * registry can run N analysts against a run and a single renderer can
482
+ * compose findings without knowing which analyzer produced them.
483
+ *
484
+ * The contract is intentionally domain-agnostic: nothing here knows
485
+ * about code, voice, RAG, or any particular agent stack. Analysts
486
+ * declare what INPUT KIND they need (a trace store, an artifact dir,
487
+ * a RunRecord, a JudgeInput, or `custom`), and the registry routes
488
+ * the matching input from `AnalystRunInputs`.
489
+ */
490
+ /**
491
+ * Compute the stable finding_id from the identity-defining fields.
492
+ * Default implementation hashes {analyst_id, area, subject, normalized claim}.
493
+ * Analysts that emit findings whose claim text varies per run (timestamps,
494
+ * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
495
+ * or (b) move the variable part into `rationale`/`metadata` and keep the
496
+ * `claim` static.
497
+ */
498
+ function computeFindingId(input) {
499
+ const basis = JSON.stringify({
500
+ a: input.analyst_id,
501
+ r: input.area,
502
+ s: input.subject ?? "",
503
+ c: normalizeClaim(input.id_basis ?? input.claim)
504
+ });
505
+ return `f_${createHash("sha256").update(basis).digest("hex").slice(0, 20)}`;
506
+ }
507
+ function normalizeClaim(c) {
508
+ return c.toLowerCase().replace(/\s+/g, " ").replace(/[.!?;:,]+$/g, "").trim();
509
+ }
510
+ /**
511
+ * Convenience factory: produce a fully-formed AnalystFinding with the
512
+ * id computed automatically. Analyst code stays terse.
513
+ */
514
+ function makeFinding(init) {
515
+ const { id_basis, produced_at, ...rest } = init;
516
+ return {
517
+ schema_version: "1.0.0",
518
+ finding_id: computeFindingId({
519
+ analyst_id: rest.analyst_id,
520
+ area: rest.area,
521
+ subject: rest.subject,
522
+ claim: rest.claim,
523
+ id_basis
524
+ }),
525
+ produced_at: produced_at ?? (/* @__PURE__ */ new Date()).toISOString(),
526
+ ...rest
527
+ };
528
+ }
529
+ //#endregion
530
+ //#region src/analyst/behavioral-analyst.ts
531
+ /**
532
+ * `behavioralAnalyst` — a DETERMINISTIC analyst (cost.kind = 'deterministic',
533
+ * never calls the LLM). It produces the efficiency/behavioral findings a
534
+ * tolerant agentic analyzer (HALO) re-derives per run inside the model —
535
+ * context bloat, output decay, tool monoculture, missing self-verification —
536
+ * directly from arithmetic over spans (`computeTraceMetrics`).
537
+ *
538
+ * Why it matters: these findings are model-agnostic BY CONSTRUCTION (no model
539
+ * in the loop), so they cannot return 0 on a weak model the way the Ax-RLM
540
+ * does — and they are strictly more reliable than HALO, which spends tokens
541
+ * re-deriving the same numbers and can hallucinate the trend. The agentic
542
+ * RLM kinds remain for SEMANTIC findings that genuinely need a model; this
543
+ * analyst owns the behavioral class.
544
+ */
545
+ const RECOMMENDED_ACTION = {
546
+ "monotonic-input-growth": "Inspect context assembly; if prior history is repeatedly included, summarize completed work before the next model call.",
547
+ "output-length-decay": "Check late-step completeness; if shorter responses omit required work, add explicit completion criteria to the agent instructions.",
548
+ "single-tool-dependency": "Test whether an inspect or verification tool improves outcomes after the repeated call fails or returns no progress.",
549
+ "no-self-verification": "After state-changing actions, require an observable check before the agent proceeds."
550
+ };
551
+ const ANALYST_ID = "efficiency-behavioral";
552
+ const AGGREGATE_CLAIM = {
553
+ "monotonic-input-growth": (observed, analyzed) => `${observed}/${analyzed} analyzed traces showed input tokens grow from zero to nonzero or to at least 3x their initial value across at least 3 serial model calls without a decrease.`,
554
+ "output-length-decay": (observed, analyzed) => `${observed}/${analyzed} analyzed traces showed output tokens decrease while input tokens increased monotonically across at least 3 serial model calls.`,
555
+ "single-tool-dependency": (observed, analyzed) => `${observed}/${analyzed} analyzed traces used only one named tool across at least 3 tool calls.`,
556
+ "no-self-verification": (observed, analyzed) => `${observed}/${analyzed} analyzed traces had at least 3 tool calls without a verification-named tool call.`
557
+ };
558
+ /**
559
+ * Map computed signals → structured AnalystFindings. Pure: no LLM, no clock
560
+ * dependence beyond `produced_at` (overridable for deterministic tests).
561
+ */
562
+ function deriveEfficiencyFindings(metrics, opts = {}) {
563
+ const analystId = opts.analystId ?? ANALYST_ID;
564
+ const traceId = metrics.traceId;
565
+ return metrics.signals.map((sig) => makeFinding({
566
+ analyst_id: analystId,
567
+ area: "efficiency",
568
+ subject: sig.code,
569
+ claim: sig.detail,
570
+ severity: sig.severity,
571
+ confidence: 1,
572
+ evidence_refs: [{
573
+ kind: "metric",
574
+ uri: traceId ? `metric://trace/${encodeURIComponent(traceId)}/efficiency/${sig.code}` : `metric://efficiency/${sig.code}`,
575
+ excerpt: JSON.stringify(sig.evidence)
576
+ }],
577
+ recommended_action: RECOMMENDED_ACTION[sig.code],
578
+ metadata: {
579
+ deterministic: true,
580
+ evidence: sig.evidence,
581
+ ...traceId ? { trace_id: traceId } : {}
582
+ },
583
+ id_basis: sig.code,
584
+ ...opts.producedAt ? { produced_at: opts.producedAt } : {}
585
+ }));
586
+ }
587
+ /** The deterministic behavioral/efficiency analyst (no LLM, any-model). */
588
+ function behavioralAnalyst() {
589
+ return {
590
+ id: ANALYST_ID,
591
+ description: "Deterministic behavioral/efficiency findings over OTLP spans — token-growth, output-decay, tool-monoculture, missing self-verification. Zero LLM; model-agnostic by construction.",
592
+ inputKind: "trace-store",
593
+ cost: { kind: "deterministic" },
594
+ version: "2.0.0",
595
+ async analyze(store) {
596
+ const overview = await store.getOverview();
597
+ const analyzedTraceIds = [...new Set(overview.sample_trace_ids)].sort();
598
+ const findingsById = /* @__PURE__ */ new Map();
599
+ for (const traceId of analyzedTraceIds) {
600
+ const viewed = await store.viewTrace({ trace_id: traceId });
601
+ if (viewed.trace_id !== traceId) throw new Error(`behavioralAnalyst: requested trace '${traceId}', received '${viewed.trace_id}'`);
602
+ if (!viewed.spans) throw new Error(`behavioralAnalyst: trace '${traceId}' is oversized; complete spans are required`);
603
+ const metrics = computeTraceMetrics(viewed.spans);
604
+ if (metrics.traceId !== null && metrics.traceId !== traceId) throw new Error(`behavioralAnalyst: requested trace '${traceId}', received '${metrics.traceId}'`);
605
+ for (const finding of deriveEfficiencyFindings(metrics)) {
606
+ const current = findingsById.get(finding.finding_id);
607
+ if (!current) {
608
+ findingsById.set(finding.finding_id, {
609
+ finding,
610
+ traceIds: [traceId],
611
+ evidence: [...finding.evidence_refs]
612
+ });
613
+ continue;
614
+ }
615
+ current.traceIds.push(traceId);
616
+ current.evidence.push(...finding.evidence_refs);
617
+ }
618
+ }
619
+ return [...findingsById.values()].map(({ finding, traceIds, evidence }) => ({
620
+ ...finding,
621
+ claim: AGGREGATE_CLAIM[finding.subject](traceIds.length, analyzedTraceIds.length),
622
+ rationale: `${traceIds.length}/${analyzedTraceIds.length} analyzed traces exhibited this pattern.`,
623
+ evidence_refs: evidence,
624
+ metadata: {
625
+ deterministic: true,
626
+ trace_ids: traceIds,
627
+ observed_trace_count: traceIds.length,
628
+ analyzed_trace_count: analyzedTraceIds.length
629
+ }
630
+ }));
631
+ }
632
+ };
633
+ }
634
+ //#endregion
635
+ //#region src/analyst/ax-cost-service.ts
636
+ /**
637
+ * Meter every chat call an Ax program makes through the shared paid-call ledger.
638
+ * The wrapper disables provider streaming because a stream has no complete usage
639
+ * receipt until it is consumed, while Ax's analyst output is not streamed to users.
640
+ */
641
+ function meterAxChatService(ai, options) {
642
+ assertPositiveInteger(options.maxOutputTokens, "maxOutputTokens");
643
+ const source = ai;
644
+ if (typeof source.chat !== "function") throw new TypeError("meterAxChatService: Ax service must implement chat()");
645
+ const providerChat = source.chat.bind(ai);
646
+ const chat = async (request, callOptions = {}) => {
647
+ const boundedRequest = boundOutputTokens(request, options.maxOutputTokens);
648
+ const model = modelName(boundedRequest.model) || options.defaultModel || "";
649
+ const canTurnOffThinking = canDisableThinking(ai, model);
650
+ const combined = combineSignals(options.signal, callOptions.abortSignal);
651
+ try {
652
+ const paid = await options.ledger.runPaidCall({
653
+ channel: "analyst",
654
+ phase: options.phase ?? "analyst.ax.chat",
655
+ actor: options.actor,
656
+ model,
657
+ tags: options.tags,
658
+ signal: combined.signal,
659
+ maximumCharge: maximumChargeForAxChatRequest(boundedRequest, model),
660
+ execute: async (executionSignal) => {
661
+ const providerOptions = {
662
+ ...callOptions,
663
+ abortSignal: executionSignal,
664
+ retry: {
665
+ ...callOptions.retry,
666
+ maxRetries: 0
667
+ },
668
+ stream: false,
669
+ showThoughts: false
670
+ };
671
+ if (canTurnOffThinking) providerOptions.thinkingTokenBudget = "none";
672
+ else Reflect.deleteProperty(providerOptions, "thinkingTokenBudget");
673
+ const response = await providerChat(boundedRequest, providerOptions);
674
+ if (response instanceof ReadableStream) throw new Error("meterAxChatService: provider returned a stream after stream:false");
675
+ return response;
676
+ },
677
+ receipt: (response) => costReceiptFromAxResponse(response, model)
678
+ });
679
+ if (!paid.succeeded) throw paid.error;
680
+ return paid.value;
681
+ } finally {
682
+ combined.dispose();
683
+ }
684
+ };
685
+ return new Proxy(ai, { get(target, property) {
686
+ if (property === "chat") return chat;
687
+ const value = Reflect.get(target, property, target);
688
+ return typeof value === "function" ? value.bind(target) : value;
689
+ } });
690
+ }
691
+ /** Conservative priced bound for one Ax text chat request. */
692
+ function maximumChargeForAxChatRequest(request, defaultModel) {
693
+ const model = modelName(request.model) || defaultModel || "";
694
+ const maxTokens = request.modelConfig?.maxTokens;
695
+ const completions = request.modelConfig?.n;
696
+ if (!model || maxTokens === void 0 || completions === void 0) return void 0;
697
+ assertPositiveInteger(maxTokens, "request.modelConfig.maxTokens");
698
+ assertPositiveInteger(completions, "request.modelConfig.n");
699
+ const maximumOutputTokens = maxTokens * completions;
700
+ assertPositiveInteger(maximumOutputTokens, "maximum output tokens");
701
+ if (containsUnboundedOrCacheableContent(request)) return void 0;
702
+ let inputTokens;
703
+ try {
704
+ const pricedRequest = request.model === void 0 ? {
705
+ ...request,
706
+ model
707
+ } : request;
708
+ inputTokens = new TextEncoder().encode(JSON.stringify(pricedRequest)).byteLength * completions;
709
+ } catch {
710
+ return;
711
+ }
712
+ assertPositiveInteger(inputTokens, "maximum input tokens");
713
+ return {
714
+ model,
715
+ inputTokens,
716
+ outputTokens: maximumOutputTokens
717
+ };
718
+ }
719
+ function boundOutputTokens(request, limit) {
720
+ const requested = request.modelConfig?.maxTokens;
721
+ if (requested !== void 0) assertPositiveInteger(requested, "request.modelConfig.maxTokens");
722
+ const completions = request.modelConfig?.n ?? 1;
723
+ assertPositiveInteger(completions, "request.modelConfig.n");
724
+ const maxTokens = requested === void 0 ? limit : Math.min(requested, limit);
725
+ return {
726
+ ...request,
727
+ modelConfig: {
728
+ ...request.modelConfig,
729
+ maxTokens,
730
+ n: completions
731
+ }
732
+ };
733
+ }
734
+ function costReceiptFromAxResponse(response, fallbackModel) {
735
+ const usage = response.modelUsage;
736
+ const tokens = usage?.tokens;
737
+ const model = usage?.model || fallbackModel;
738
+ if (!tokens || !validUsage(tokens.promptTokens) || !validUsage(tokens.completionTokens) || !validUsage(tokens.totalTokens) || tokens.totalTokens < tokens.promptTokens + tokens.completionTokens || !validOptionalUsage(tokens.cacheReadTokens) || !validOptionalUsage(tokens.cacheCreationTokens) || !validOptionalUsage(tokens.reasoningTokens) || !validOptionalUsage(tokens.thoughtsTokens)) return {
739
+ model,
740
+ inputTokens: 0,
741
+ outputTokens: 0,
742
+ usageUnknown: true
743
+ };
744
+ const cacheReadTokens = validUsage(tokens.cacheReadTokens) ? tokens.cacheReadTokens : 0;
745
+ const cacheCreationTokens = validUsage(tokens.cacheCreationTokens) ? tokens.cacheCreationTokens : 0;
746
+ const totalCacheTokens = cacheReadTokens + cacheCreationTokens;
747
+ const thoughtsTokens = validUsage(tokens.thoughtsTokens) ? tokens.thoughtsTokens : 0;
748
+ const reasoningTokens = Math.max(validUsage(tokens.reasoningTokens) ? tokens.reasoningTokens : 0, thoughtsTokens);
749
+ const separateReasoningTokens = reasoningTokens > tokens.completionTokens ? reasoningTokens : 0;
750
+ const additionalOutputTokens = Math.max(thoughtsTokens, separateReasoningTokens);
751
+ const outputTokens = tokens.completionTokens + additionalOutputTokens;
752
+ const directTotal = tokens.promptTokens + tokens.completionTokens;
753
+ if (!(/* @__PURE__ */ new Set([
754
+ directTotal,
755
+ directTotal + totalCacheTokens,
756
+ directTotal + additionalOutputTokens,
757
+ directTotal + totalCacheTokens + additionalOutputTokens
758
+ ])).has(tokens.totalTokens)) return {
759
+ model,
760
+ inputTokens: 0,
761
+ outputTokens: 0,
762
+ usageUnknown: true
763
+ };
764
+ return {
765
+ model,
766
+ inputTokens: tokens.promptTokens,
767
+ outputTokens,
768
+ ...reasoningTokens > 0 ? { reasoningTokens } : {},
769
+ ...cacheReadTokens > 0 ? { cachedTokens: cacheReadTokens } : {},
770
+ ...cacheCreationTokens > 0 ? { cacheWriteTokens: cacheCreationTokens } : {}
771
+ };
772
+ }
773
+ function combineSignals(first, second) {
774
+ if (!first) return {
775
+ signal: second,
776
+ dispose: () => {}
777
+ };
778
+ if (!second || first === second) return {
779
+ signal: first,
780
+ dispose: () => {}
781
+ };
782
+ if (typeof AbortSignal.any === "function") return {
783
+ signal: AbortSignal.any([first, second]),
784
+ dispose: () => {}
785
+ };
786
+ const controller = new AbortController();
787
+ const dispose = () => {
788
+ first.removeEventListener("abort", abortFromFirst);
789
+ second.removeEventListener("abort", abortFromSecond);
790
+ };
791
+ const abortFrom = (source) => {
792
+ if (!controller.signal.aborted) controller.abort(source.reason);
793
+ dispose();
794
+ };
795
+ const abortFromFirst = () => abortFrom(first);
796
+ const abortFromSecond = () => abortFrom(second);
797
+ if (first.aborted) abortFrom(first);
798
+ else if (second.aborted) abortFrom(second);
799
+ else {
800
+ first.addEventListener("abort", abortFromFirst, { once: true });
801
+ second.addEventListener("abort", abortFromSecond, { once: true });
802
+ }
803
+ return {
804
+ signal: controller.signal,
805
+ dispose
806
+ };
807
+ }
808
+ function containsUnboundedOrCacheableContent(request) {
809
+ if (request.functions?.some((fn) => fn.cache === true)) return true;
810
+ return request.chatPrompt.some((message) => {
811
+ if (message.cache === true) return true;
812
+ if (!("content" in message) || !Array.isArray(message.content)) return false;
813
+ return message.content.some((part) => part.cache === true || part.type !== "text");
814
+ });
815
+ }
816
+ function modelName(value) {
817
+ return typeof value === "string" ? value : "";
818
+ }
819
+ function canDisableThinking(ai, model) {
820
+ const namedAi = ai;
821
+ const serviceName = typeof namedAi.getName === "function" ? namedAi.getName() : "";
822
+ const modelId = model.slice(model.lastIndexOf("/") + 1);
823
+ return serviceName !== "GoogleGeminiAI" || !/^gemini-3(?:[.-]|$)/i.test(modelId);
824
+ }
825
+ function validUsage(value) {
826
+ return typeof value === "number" && Number.isSafeInteger(value) && value >= 0;
827
+ }
828
+ function validOptionalUsage(value) {
829
+ return value === void 0 || validUsage(value);
830
+ }
831
+ function assertPositiveInteger(value, field) {
832
+ if (!Number.isSafeInteger(value) || value <= 0) throw new RangeError(`meterAxChatService: ${field} must be a positive integer`);
833
+ }
834
+ //#endregion
835
+ //#region src/analyst/finding-subject.ts
836
+ /**
837
+ * Typed `FindingSubject` — the canonical grammar every analyst kind emits.
838
+ *
839
+ * Background: kind actor prompts have always documented a subject grammar
840
+ * (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`) but the
841
+ * LLM was unconstrained — it could emit `subject: "fix the prompt"`
842
+ * (prose) and downstream adapters routed on `startsWith(...)` would
843
+ * silently skip it. Every per-vertical `ImprovementAdapter` had a
844
+ * routing table that mostly caught nothing.
845
+ *
846
+ * This module fixes that:
847
+ * - `parseFindingSubject(raw)` — returns the typed `FindingSubject`
848
+ * when `raw` matches the grammar, else `null`. Used at the
849
+ * `RawAnalystFindingSchema` boundary so malformed subjects are
850
+ * rejected loudly instead of silently lifted into the registry.
851
+ * - `FindingSubjectKind` — the union of valid locus categories. Each
852
+ * variant carries the typed components downstream adapters resolve
853
+ * against the agent's surface manifest (no string parsing in the
854
+ * adapter).
855
+ * - `FINDING_SUBJECT_GRAMMAR_PROMPT` — single source of truth for the
856
+ * grammar string embedded in kind actor prompts. Drift between
857
+ * prompt and parser is impossible if every kind imports this.
858
+ *
859
+ * The grammar is intentionally NARROW — only loci the substrate's
860
+ * default `ImprovementAdapter` / `KnowledgeAdapter` can act on. A
861
+ * finding with a subject outside this set fails the parser; the kind
862
+ * author either extends the grammar here (and adds adapter routing)
863
+ * or rephrases the prompt to map onto an existing variant.
864
+ *
865
+ * `failure-mode` is the one exception — its subjects are free-form
866
+ * cluster labels, not loci. The schema preserves them as
867
+ * `{ kind: 'cluster', label }` and the adapters skip them (cluster
868
+ * findings are evidence, not actionable mutations).
869
+ */
870
+ const FINDING_SUBJECT_KINDS = [
871
+ "knowledge.wiki",
872
+ "knowledge.claim",
873
+ "knowledge.raw",
874
+ "knowledge.stale",
875
+ "system-prompt",
876
+ "skill",
877
+ "tool-doc",
878
+ "new-tool",
879
+ "mcp",
880
+ "hook",
881
+ "subagent",
882
+ "workflow",
883
+ "rollout-policy",
884
+ "agent-profile",
885
+ "code",
886
+ "rag",
887
+ "memory",
888
+ "scaffolding",
889
+ "output-schema",
890
+ "websearch.outdated",
891
+ "prior-run-summary",
892
+ "cluster"
893
+ ];
894
+ /**
895
+ * Parse a raw subject string emitted by an analyst kind's actor.
896
+ *
897
+ * Returns the typed `FindingSubject` when `raw` matches the grammar,
898
+ * else `null`. Callers use the `null` return as a signal to either
899
+ * (a) reject the finding at parse time (kinds that emit typed loci —
900
+ * knowledge-gap, improvement, knowledge-poisoning) or (b) lift it as
901
+ * a cluster label (failure-mode).
902
+ *
903
+ * Slugs are constrained to `[a-z0-9-]+` (lowercase kebab) to keep file
904
+ * paths sane downstream. Topics / keys / sections allow any non-empty
905
+ * string (free-form for the LLM's voice) but get trimmed.
906
+ *
907
+ * Empty / whitespace-only inputs return `null`. `undefined` returns
908
+ * `null`. Both are surfaced by the caller as a rejected subject.
909
+ */
910
+ function parseFindingSubject(raw) {
911
+ if (raw === null || raw === void 0) return null;
912
+ const trimmed = raw.trim();
913
+ if (trimmed.length === 0) return null;
914
+ const wiki = trimmed.match(/^agent-knowledge:wiki:([a-z0-9][a-z0-9-]*)(?:#([a-z0-9][a-z0-9-]*))?$/);
915
+ if (wiki) return {
916
+ kind: "knowledge.wiki",
917
+ slug: wiki[1],
918
+ ...wiki[2] ? { heading: wiki[2] } : {}
919
+ };
920
+ const claim = trimmed.match(/^agent-knowledge:claim:(.+)$/);
921
+ if (claim && claim[1].trim().length > 0) return {
922
+ kind: "knowledge.claim",
923
+ topic: claim[1].trim()
924
+ };
925
+ const raw_ = trimmed.match(/^agent-knowledge:raw:(.+)$/);
926
+ if (raw_ && raw_[1].trim().length > 0) return {
927
+ kind: "knowledge.raw",
928
+ sourceId: raw_[1].trim()
929
+ };
930
+ const stale = trimmed.match(/^agent-knowledge:stale:([a-z0-9][a-z0-9-]*)$/);
931
+ if (stale) return {
932
+ kind: "knowledge.stale",
933
+ slug: stale[1]
934
+ };
935
+ const sp = trimmed.match(/^system-prompt:(.+)$/);
936
+ if (sp && sp[1].trim().length > 0) return {
937
+ kind: "system-prompt",
938
+ section: sp[1].trim()
939
+ };
940
+ const skill = trimmed.match(/^skill:([a-z0-9][a-z0-9_.-]*)$/);
941
+ if (skill) return {
942
+ kind: "skill",
943
+ name: skill[1]
944
+ };
945
+ const tdAspect = trimmed.match(/^tool-doc:([a-z0-9][a-z0-9_-]*):(.+)$/);
946
+ if (tdAspect && tdAspect[2].trim().length > 0) return {
947
+ kind: "tool-doc",
948
+ tool: tdAspect[1],
949
+ aspect: tdAspect[2].trim()
950
+ };
951
+ const td = trimmed.match(/^tool-doc:([a-z0-9][a-z0-9_-]*)$/);
952
+ if (td) return {
953
+ kind: "tool-doc",
954
+ tool: td[1]
955
+ };
956
+ const nt = trimmed.match(/^new-tool:([a-z0-9][a-z0-9_-]*)$/);
957
+ if (nt) return {
958
+ kind: "new-tool",
959
+ name: nt[1]
960
+ };
961
+ const mcp = trimmed.match(/^mcp:([a-z0-9][a-z0-9_.-]*)(?::([a-z0-9][a-z0-9_.-]*))?$/);
962
+ if (mcp) return {
963
+ kind: "mcp",
964
+ server: mcp[1],
965
+ ...mcp[2] ? { tool: mcp[2] } : {}
966
+ };
967
+ const hook = trimmed.match(/^hook:([a-z0-9][a-z0-9_.-]*)$/);
968
+ if (hook) return {
969
+ kind: "hook",
970
+ name: hook[1]
971
+ };
972
+ const subagent = trimmed.match(/^subagent:([a-z0-9][a-z0-9_.-]*)$/);
973
+ if (subagent) return {
974
+ kind: "subagent",
975
+ name: subagent[1]
976
+ };
977
+ const workflow = trimmed.match(/^workflow:([a-z0-9][a-z0-9_.-]*)$/);
978
+ if (workflow) return {
979
+ kind: "workflow",
980
+ name: workflow[1]
981
+ };
982
+ const rolloutPolicy = trimmed.match(/^rollout-policy:(.+)$/);
983
+ if (rolloutPolicy && rolloutPolicy[1].trim().length > 0) return {
984
+ kind: "rollout-policy",
985
+ field: rolloutPolicy[1].trim()
986
+ };
987
+ const agentProfile = trimmed.match(/^agent-profile:(.+)$/);
988
+ if (agentProfile && agentProfile[1].trim().length > 0) return {
989
+ kind: "agent-profile",
990
+ field: agentProfile[1].trim()
991
+ };
992
+ const code = trimmed.match(/^code:(.+)$/);
993
+ if (code && code[1].trim().length > 0) return {
994
+ kind: "code",
995
+ path: code[1].trim()
996
+ };
997
+ const rag = trimmed.match(/^rag:([a-z0-9][a-z0-9_-]*):(.+)$/);
998
+ if (rag && rag[2].trim().length > 0) return {
999
+ kind: "rag",
1000
+ corpus: rag[1],
1001
+ docId: rag[2].trim()
1002
+ };
1003
+ const mem = trimmed.match(/^memory:(.+)$/);
1004
+ if (mem && mem[1].trim().length > 0) return {
1005
+ kind: "memory",
1006
+ key: mem[1].trim()
1007
+ };
1008
+ const sc = trimmed.match(/^scaffolding:(.+)$/);
1009
+ if (sc && sc[1].trim().length > 0) return {
1010
+ kind: "scaffolding",
1011
+ concern: sc[1].trim()
1012
+ };
1013
+ const os = trimmed.match(/^output-schema:(.+)$/);
1014
+ if (os && os[1].trim().length > 0) return {
1015
+ kind: "output-schema",
1016
+ field: os[1].trim()
1017
+ };
1018
+ const ws = trimmed.match(/^websearch:outdated:(.+)$/);
1019
+ if (ws && ws[1].trim().length > 0) return {
1020
+ kind: "websearch.outdated",
1021
+ topic: ws[1].trim()
1022
+ };
1023
+ const prs = trimmed.match(/^prior-run-summary:(.+)$/);
1024
+ if (prs && prs[1].trim().length > 0) return {
1025
+ kind: "prior-run-summary",
1026
+ topic: prs[1].trim()
1027
+ };
1028
+ if (/^[a-z0-9][a-z0-9._-]*$/.test(trimmed) && trimmed.length <= 80) return {
1029
+ kind: "cluster",
1030
+ label: trimmed
1031
+ };
1032
+ return null;
1033
+ }
1034
+ /**
1035
+ * Render the parsed subject back to its canonical string form. Inverse
1036
+ * of `parseFindingSubject`; useful when the substrate constructs new
1037
+ * findings programmatically (e.g. for tests, replays, or
1038
+ * `id_basis` carry-forward).
1039
+ */
1040
+ function renderFindingSubject(s) {
1041
+ switch (s.kind) {
1042
+ case "knowledge.wiki": return s.heading ? `agent-knowledge:wiki:${s.slug}#${s.heading}` : `agent-knowledge:wiki:${s.slug}`;
1043
+ case "knowledge.claim": return `agent-knowledge:claim:${s.topic}`;
1044
+ case "knowledge.raw": return `agent-knowledge:raw:${s.sourceId}`;
1045
+ case "knowledge.stale": return `agent-knowledge:stale:${s.slug}`;
1046
+ case "system-prompt": return `system-prompt:${s.section}`;
1047
+ case "skill": return `skill:${s.name}`;
1048
+ case "tool-doc": return s.aspect ? `tool-doc:${s.tool}:${s.aspect}` : `tool-doc:${s.tool}`;
1049
+ case "new-tool": return `new-tool:${s.name}`;
1050
+ case "mcp": return s.tool ? `mcp:${s.server}:${s.tool}` : `mcp:${s.server}`;
1051
+ case "hook": return `hook:${s.name}`;
1052
+ case "subagent": return `subagent:${s.name}`;
1053
+ case "workflow": return `workflow:${s.name}`;
1054
+ case "rollout-policy": return `rollout-policy:${s.field}`;
1055
+ case "agent-profile": return `agent-profile:${s.field}`;
1056
+ case "code": return `code:${s.path}`;
1057
+ case "rag": return `rag:${s.corpus}:${s.docId}`;
1058
+ case "memory": return `memory:${s.key}`;
1059
+ case "scaffolding": return `scaffolding:${s.concern}`;
1060
+ case "output-schema": return `output-schema:${s.field}`;
1061
+ case "websearch.outdated": return `websearch:outdated:${s.topic}`;
1062
+ case "prior-run-summary": return `prior-run-summary:${s.topic}`;
1063
+ case "cluster": return s.label;
1064
+ }
1065
+ }
1066
+ /**
1067
+ * The grammar text embedded into kind actor prompts. Kinds opt into
1068
+ * the subset of variants they emit (e.g. `improvement` excludes the
1069
+ * cluster variant; `failure-mode` includes ONLY the cluster variant).
1070
+ *
1071
+ * Drift between prompt and parser is impossible: every kind imports
1072
+ * this constant + the matching `expects` set, and the unit tests below
1073
+ * lock the table to the parser.
1074
+ */
1075
+ const FINDING_SUBJECT_SYNTAX = {
1076
+ "knowledge.wiki": "agent-knowledge:wiki:<slug>[#<heading>]",
1077
+ "knowledge.claim": "agent-knowledge:claim:<topic>",
1078
+ "knowledge.raw": "agent-knowledge:raw:<source-id>",
1079
+ "knowledge.stale": "agent-knowledge:stale:<slug>",
1080
+ "system-prompt": "system-prompt:<section>",
1081
+ skill: "skill:<name>",
1082
+ "tool-doc": "tool-doc:<tool>[:<aspect>]",
1083
+ "new-tool": "new-tool:<name>",
1084
+ mcp: "mcp:<server>[:<tool>]",
1085
+ hook: "hook:<name>",
1086
+ subagent: "subagent:<name>",
1087
+ workflow: "workflow:<name>",
1088
+ "rollout-policy": "rollout-policy:<field>",
1089
+ "agent-profile": "agent-profile:<field>",
1090
+ code: "code:<path>",
1091
+ rag: "rag:<corpus>:<doc-id>",
1092
+ memory: "memory:<key>",
1093
+ scaffolding: "scaffolding:<concern>",
1094
+ "output-schema": "output-schema:<field>",
1095
+ "websearch.outdated": "websearch:outdated:<topic>",
1096
+ "prior-run-summary": "prior-run-summary:<topic>",
1097
+ cluster: "<lowercase-cluster-label>"
1098
+ };
1099
+ const FINDING_SUBJECT_PURPOSE = {
1100
+ "knowledge.wiki": "create or update a wiki page",
1101
+ "knowledge.claim": "draft a claim or relation",
1102
+ "knowledge.raw": "curate a raw source",
1103
+ "knowledge.stale": "mark a stale page",
1104
+ "system-prompt": "revise a system-prompt section",
1105
+ skill: "create or revise a skill",
1106
+ "tool-doc": "revise a tool contract",
1107
+ "new-tool": "propose a new tool",
1108
+ mcp: "revise an MCP server or tool",
1109
+ hook: "revise a lifecycle hook",
1110
+ subagent: "revise a delegated agent",
1111
+ workflow: "revise an orchestration workflow",
1112
+ "rollout-policy": "revise budget, sampling, or stop policy",
1113
+ "agent-profile": "revise another AgentProfile field",
1114
+ code: "revise an implementation path",
1115
+ rag: "ingest or correct a RAG document",
1116
+ memory: "invalidate or set memory",
1117
+ scaffolding: "revise preconditions, retries, or verification",
1118
+ "output-schema": "constrain the output shape",
1119
+ "websearch.outdated": "identify a stale web result",
1120
+ "prior-run-summary": "identify a stale prior-run summary",
1121
+ cluster: "name one failure cluster"
1122
+ };
1123
+ function renderFindingSubjectGrammar(kinds) {
1124
+ return [
1125
+ "Subjects MUST match one of these forms — anything else is rejected at parse time:",
1126
+ ...kinds.map((kind) => ` ${FINDING_SUBJECT_SYNTAX[kind]} — ${FINDING_SUBJECT_PURPOSE[kind]}`),
1127
+ "Runtime ids are lowercase [a-z0-9_.-]+. Topics, keys, paths, and sections are free-form and trimmed."
1128
+ ].join("\n");
1129
+ }
1130
+ const FINDING_SUBJECT_GRAMMAR_PROMPT = renderFindingSubjectGrammar(FINDING_SUBJECT_KINDS);
1131
+ /**
1132
+ * The variants each kind is allowed to emit. Used at the kind factory
1133
+ * boundary so a knowledge-gap finding can't sneak in a `system-prompt:*`
1134
+ * subject (the improvement-analyst's job) and vice versa.
1135
+ *
1136
+ * `failure-mode` is restricted to `cluster` — the only kind that emits
1137
+ * a non-locus subject.
1138
+ */
1139
+ const KIND_EXPECTED_SUBJECTS = {
1140
+ "failure-mode": ["cluster"],
1141
+ "knowledge-gap": [
1142
+ "knowledge.wiki",
1143
+ "knowledge.claim",
1144
+ "knowledge.raw",
1145
+ "knowledge.stale",
1146
+ "tool-doc",
1147
+ "system-prompt",
1148
+ "skill",
1149
+ "mcp",
1150
+ "subagent",
1151
+ "workflow",
1152
+ "memory",
1153
+ "websearch.outdated",
1154
+ "prior-run-summary"
1155
+ ],
1156
+ "knowledge-poisoning": [
1157
+ "knowledge.wiki",
1158
+ "knowledge.claim",
1159
+ "knowledge.raw",
1160
+ "tool-doc",
1161
+ "system-prompt",
1162
+ "skill",
1163
+ "mcp",
1164
+ "hook",
1165
+ "memory",
1166
+ "websearch.outdated",
1167
+ "prior-run-summary"
1168
+ ],
1169
+ improvement: [
1170
+ "system-prompt",
1171
+ "skill",
1172
+ "tool-doc",
1173
+ "new-tool",
1174
+ "mcp",
1175
+ "hook",
1176
+ "subagent",
1177
+ "workflow",
1178
+ "rollout-policy",
1179
+ "agent-profile",
1180
+ "code",
1181
+ "rag",
1182
+ "memory",
1183
+ "scaffolding",
1184
+ "output-schema",
1185
+ "knowledge.wiki",
1186
+ "knowledge.claim"
1187
+ ]
1188
+ };
1189
+ /** Render only the subject forms one analyst kind is permitted to emit. */
1190
+ function findingSubjectGrammarPromptFor(kindId) {
1191
+ const kinds = KIND_EXPECTED_SUBJECTS[kindId];
1192
+ if (!kinds) throw new Error(`unknown analyst kind: ${kindId}`);
1193
+ return renderFindingSubjectGrammar(kinds);
1194
+ }
1195
+ /**
1196
+ * Zod schema that validates a raw subject string and returns the parsed
1197
+ * `FindingSubject`. Embedded in `RawAnalystFindingSchema` via
1198
+ * `transform`, so `subject` arrives at the kind factory either as a
1199
+ * typed locus or as a parse error attached to a single Zod issue.
1200
+ *
1201
+ * Optionality is preserved: subjects ARE optional on the wire (some
1202
+ * findings are descriptive, not actionable). When present, they MUST
1203
+ * parse — emitting a malformed subject is a contract violation, not a
1204
+ * soft signal.
1205
+ */
1206
+ const FindingSubjectStringSchema = z.string().refine((s) => parseFindingSubject(s) !== null, { message: "subject does not match the finding-subject grammar" });
1207
+ //#endregion
1208
+ //#region src/analyst/parse-tolerant.ts
1209
+ /**
1210
+ * Forgiving pre-parse for analyst findings. Weak models routinely emit
1211
+ * schema-correct content in an unusable wrapper — fenced ```json blocks, a
1212
+ * single object where an array is expected, trailing commas. Measured: GPT-4o
1213
+ * drops to 0% usable output purely from markdown-fence wrapping
1214
+ * (arXiv:2605.02363). A five-line de-fence recovers most of it. This module is
1215
+ * the de-fence/coerce step that runs BEFORE Zod, so a recoverable finding is
1216
+ * repaired, not dropped.
1217
+ *
1218
+ * Pure + deterministic. No model, no network.
1219
+ */
1220
+ /** Strip a ```lang ... ``` (or bare ``` ... ```) code fence, if the string is one. */
1221
+ function stripCodeFences(text) {
1222
+ const t = text.trim();
1223
+ const m = t.match(/^```[a-zA-Z0-9]*\s*\n?([\s\S]*?)\n?```$/);
1224
+ return m ? m[1].trim() : t;
1225
+ }
1226
+ /** Remove trailing commas before } or ] — the most common near-JSON defect. */
1227
+ function dropTrailingCommas(s) {
1228
+ return s.replace(/,(\s*[}\]])/g, "$1");
1229
+ }
1230
+ /**
1231
+ * Best-effort parse of a string into JSON. De-fences, drops trailing commas,
1232
+ * then `JSON.parse`. Returns `undefined` (never throws) when unrecoverable.
1233
+ */
1234
+ function coerceJson(text) {
1235
+ const candidate = dropTrailingCommas(stripCodeFences(text));
1236
+ try {
1237
+ return JSON.parse(candidate);
1238
+ } catch {
1239
+ return;
1240
+ }
1241
+ }
1242
+ /**
1243
+ * Coerce arbitrary actor/structurer output into an array of candidate finding
1244
+ * rows: a JSON string → parse; a single object → 1-element array; an array →
1245
+ * as-is; anything else → []. Callers still run each row through Zod
1246
+ * (`parseRawFinding`) — this only fixes the shape and never invents fields.
1247
+ */
1248
+ function coerceToFindingRows(raw) {
1249
+ let value = raw;
1250
+ if (typeof value === "string") {
1251
+ const parsed = coerceJson(value);
1252
+ if (parsed === void 0) return [];
1253
+ value = parsed;
1254
+ }
1255
+ if (Array.isArray(value)) return value;
1256
+ if (value && typeof value === "object") {
1257
+ const inner = value.findings;
1258
+ if (Array.isArray(inner)) return inner;
1259
+ return [value];
1260
+ }
1261
+ return [];
1262
+ }
1263
+ //#endregion
1264
+ //#region src/analyst/finding-signature.ts
1265
+ /**
1266
+ * Typed Ax output for analyst findings.
1267
+ *
1268
+ * Ax binds the field as `findings:json[]` so the provider emits native
1269
+ * structured output. At the kind-factory boundary every row is validated
1270
+ * before it becomes an `AnalystFinding`.
1271
+ *
1272
+ * Why not `f.object().array()` directly in the signature? The Ax
1273
+ * signature string `question:string -> findings:json[]` already lets
1274
+ * the provider emit JSON arrays. A Zod boundary is required either
1275
+ * way (the provider can return any JSON), and Zod gives us a single
1276
+ * validation surface independent of which Ax version is installed.
1277
+ */
1278
+ const ANALYST_SEVERITIES = [
1279
+ "critical",
1280
+ "high",
1281
+ "medium",
1282
+ "low",
1283
+ "info"
1284
+ ];
1285
+ const RawAnalystEvidenceSchema = z.object({
1286
+ uri: z.string().trim().min(1).max(2e3),
1287
+ excerpt: z.string().max(2e3).optional()
1288
+ }).strict();
1289
+ const RawAnalystFindingBaseShape = {
1290
+ severity: z.enum(ANALYST_SEVERITIES),
1291
+ claim: z.string().min(1).max(2e3),
1292
+ subject: z.string().max(400).refine((subject) => parseFindingSubject(subject) !== null, { message: "subject does not match the finding-subject grammar" }).optional(),
1293
+ confidence: z.number().min(0).max(1),
1294
+ rationale: z.string().max(4e3).optional(),
1295
+ recommended_action: z.string().max(2e3).optional()
1296
+ };
1297
+ const RawAnalystFindingSchema = z.object({
1298
+ ...RawAnalystFindingBaseShape,
1299
+ evidence: z.array(RawAnalystEvidenceSchema).min(1)
1300
+ }).strict();
1301
+ /**
1302
+ * Description embedded into the actor prompt so the LLM knows what
1303
+ * shape to emit. Kept here so kinds share one source of truth rather
1304
+ * than restating the schema in every prompt.
1305
+ */
1306
+ const RAW_FINDING_SCHEMA_PROMPT = `Each finding MUST be a strict JSON object with:
1307
+ - severity: "critical" | "high" | "medium" | "low" | "info"
1308
+ - claim: one-sentence statement (max 2000 chars)
1309
+ - subject?: one exact subject form listed by this kind; omit rather than guess
1310
+ - evidence: REQUIRED non-empty array of {"uri": string, "excerpt"?: string}. Use real identifiers with span://, event://, artifact://, metric://, or finding://. Include a short exact quote in excerpt when available. If nothing is citable, do not emit the finding.
1311
+ - confidence: number 0..1 (0.9+ exact evidence; 0.6-0.8 inferred pattern; <0.5 speculative)
1312
+ - rationale?: one or two reasoning sentences
1313
+ - recommended_action?: concrete imperative change; omit for descriptive findings
1314
+
1315
+ Unknown fields are rejected. Do not emit area; the factory assigns it. Emit [] when there are no findings. Never fabricate evidence.`;
1316
+ /** Convert raw citations into the public finding evidence envelope. */
1317
+ function evidenceRefsFromRawFinding(finding) {
1318
+ return finding.evidence.map(({ uri, excerpt }) => ({
1319
+ kind: evidenceKindFromUri(uri),
1320
+ uri,
1321
+ excerpt
1322
+ }));
1323
+ }
1324
+ function parseRawFinding(row, log) {
1325
+ return parseFindingWithSchema(RawAnalystFindingSchema, row, log);
1326
+ }
1327
+ function parseFindingWithSchema(schema, row, log) {
1328
+ const result = schema.safeParse(row);
1329
+ if (result.success) return result.data;
1330
+ if (typeof row === "string") {
1331
+ const coerced = coerceJson(row);
1332
+ if (coerced !== void 0) {
1333
+ const retry = schema.safeParse(coerced);
1334
+ if (retry.success) return retry.data;
1335
+ }
1336
+ }
1337
+ log?.("finding rejected: schema failure", { issues: result.error.issues.map((i) => ({
1338
+ path: i.path.join("."),
1339
+ code: i.code,
1340
+ message: i.message
1341
+ })) });
1342
+ return null;
1343
+ }
1344
+ function evidenceKindFromUri(uri) {
1345
+ if (uri.startsWith("span://")) return "span";
1346
+ if (uri.startsWith("event://")) return "event";
1347
+ if (uri.startsWith("finding://")) return "finding";
1348
+ if (uri.startsWith("metric://")) return "metric";
1349
+ return "artifact";
1350
+ }
1351
+ //#endregion
1352
+ //#region src/analyst/structure-findings.ts
1353
+ /**
1354
+ * `structureFindings` — the deferred structuring pass (DSPy TwoStepAdapter /
1355
+ * HALO `synthesize_traces` analog). The agentic actor reasons FREE-FORM and
1356
+ * emits a prose `report` (which any model does reliably); this separate, cheap
1357
+ * call's ONLY job is to turn that report into `AnalystFinding[]`. Decoupling
1358
+ * reasoning from structuring is what makes the SEMANTIC findings model-agnostic
1359
+ * — the reasoning model never has to satisfy a strict typed-array contract
1360
+ * while it diagnoses.
1361
+ *
1362
+ * Forgiving: the response runs through `coerceToFindingRows` (de-fence, lift
1363
+ * single→array) before Zod, and on a zero-finding extraction from a substantive
1364
+ * report it reasks ONCE with the schema restated. Returns a typed outcome so a
1365
+ * legitimate "nothing to report" is distinguishable from a failed extraction
1366
+ * (no silent empty).
1367
+ */
1368
+ const SYSTEM = [
1369
+ "You convert a free-form trace-analysis report into a STRICT JSON array of findings.",
1370
+ "Output ONLY the JSON array — no prose, no code fences.",
1371
+ RAW_FINDING_SCHEMA_PROMPT,
1372
+ "Omit subject when the report does not contain an exact valid locus.",
1373
+ "If the report asserts NO problems, output exactly []."
1374
+ ].join(" ");
1375
+ function buildRows(raw, opts) {
1376
+ const rows = coerceToFindingRows(raw);
1377
+ const out = [];
1378
+ for (const row of rows) {
1379
+ const parsed = parseRawFinding(row);
1380
+ if (!parsed) continue;
1381
+ const callbackResult = opts.processRow ? opts.processRow(parsed) : parsed;
1382
+ if (!callbackResult) continue;
1383
+ const processed = parseRawFinding(callbackResult);
1384
+ if (!processed) continue;
1385
+ out.push(makeFinding({
1386
+ analyst_id: opts.analystId,
1387
+ area: opts.area,
1388
+ subject: processed.subject,
1389
+ claim: processed.claim,
1390
+ rationale: processed.rationale,
1391
+ severity: processed.severity,
1392
+ confidence: processed.confidence,
1393
+ evidence_refs: evidenceRefsFromRawFinding(processed),
1394
+ recommended_action: processed.recommended_action,
1395
+ ...opts.findingMetadata ? { metadata: { ...opts.findingMetadata } } : {}
1396
+ }));
1397
+ }
1398
+ return out;
1399
+ }
1400
+ async function structureFindings(opts) {
1401
+ const maxReasks = opts.maxReasks ?? 1;
1402
+ if (!Number.isSafeInteger(maxReasks) || maxReasks < 0) throw new RangeError("structureFindings: maxReasks must be a non-negative safe integer");
1403
+ const llm = {
1404
+ baseUrl: opts.baseUrl,
1405
+ apiKey: opts.apiKey,
1406
+ fetch: opts.fetchImpl
1407
+ };
1408
+ const costLedger = opts.costLedger ?? new CostLedger();
1409
+ let user = `TRACE-ANALYSIS REPORT:\n${opts.report}\n\nReturn the findings JSON array.`;
1410
+ for (let attempt = 0; attempt <= maxReasks; attempt++) {
1411
+ const request = {
1412
+ model: opts.model,
1413
+ messages: [{
1414
+ role: "system",
1415
+ content: SYSTEM
1416
+ }, {
1417
+ role: "user",
1418
+ content: user
1419
+ }],
1420
+ maxTokens: opts.maxTokens ?? 2e3
1421
+ };
1422
+ const paid = await costLedger.runPaidCall({
1423
+ channel: "analyst",
1424
+ phase: opts.costPhase ?? "analyst.structure-findings",
1425
+ actor: "structure-findings",
1426
+ model: opts.model,
1427
+ signal: opts.signal,
1428
+ maximumCharge: maximumChargeForLlmRequest(request, llm),
1429
+ tags: {
1430
+ ...opts.costTags,
1431
+ analystId: opts.analystId,
1432
+ attempt: String(attempt)
1433
+ },
1434
+ execute: (signal, callId) => callLlm(request, {
1435
+ ...llm,
1436
+ signal,
1437
+ idempotencyKey: callId
1438
+ }),
1439
+ receipt: costReceiptFromLlm,
1440
+ receiptFromError: costReceiptFromLlmError
1441
+ });
1442
+ if (!paid.succeeded) throw paid.error;
1443
+ const findings = buildRows(paid.value.content.trim(), opts);
1444
+ if (findings.length > 0) return {
1445
+ findings,
1446
+ outcome: "ok"
1447
+ };
1448
+ if (opts.report.trim().length < 200) return {
1449
+ findings: [],
1450
+ outcome: "ok"
1451
+ };
1452
+ user = `${user}\n\nThat produced no valid findings. The report DOES describe issues — re-extract them as the strict JSON array described in the system prompt. Output ONLY the array.`;
1453
+ }
1454
+ return {
1455
+ findings: [],
1456
+ outcome: "extraction_failed"
1457
+ };
1458
+ }
1459
+ /** Convert one ledger channel's complete call set into one analyst receipt. */
1460
+ function usageReceiptFromCostLedger(ledger, filter = "analyst") {
1461
+ const resolvedFilter = typeof filter === "string" ? { channel: filter } : filter;
1462
+ const summary = ledger.summary(resolvedFilter);
1463
+ const receipts = ledger.list(resolvedFilter);
1464
+ const hasReasoningUsage = receipts.some((receipt) => receipt.reasoningTokens !== void 0);
1465
+ const hasCacheWriteUsage = receipts.some((receipt) => receipt.cacheWriteTokens !== void 0);
1466
+ const cost = summary.pendingCalls > 0 || receipts.some((receipt) => receipt.costUnknown) ? {
1467
+ kind: "uncaptured",
1468
+ usd: null
1469
+ } : receipts.every((receipt) => receipt.actualCostUsd !== void 0) ? {
1470
+ kind: "observed",
1471
+ usd: summary.totalCostUsd
1472
+ } : {
1473
+ kind: "estimated",
1474
+ usd: summary.totalCostUsd
1475
+ };
1476
+ return {
1477
+ calls: summary.totalCalls + summary.pendingCalls,
1478
+ tokens: summary.usageComplete ? {
1479
+ input: summary.inputTokens,
1480
+ output: summary.outputTokens,
1481
+ ...hasReasoningUsage ? { reasoning: summary.reasoningTokens ?? 0 } : {},
1482
+ ...summary.cachedTokens > 0 ? { cached: summary.cachedTokens } : {},
1483
+ ...hasCacheWriteUsage ? { cacheWrite: summary.cacheWriteTokens ?? 0 } : {}
1484
+ } : null,
1485
+ cost,
1486
+ ...cost.kind === "uncaptured" ? { knownCostUsd: summary.totalCostUsd } : {}
1487
+ };
1488
+ }
1489
+ /** Wait a bounded time for late provider receipts, then take one immutable snapshot. */
1490
+ async function settleUsageReceiptFromCostLedger(ledger, options = {}) {
1491
+ const { timeoutMs: requestedTimeoutMs, ...requestedFilter } = options;
1492
+ const filter = {
1493
+ channel: requestedFilter.channel ?? "analyst",
1494
+ ...requestedFilter.phase === void 0 ? {} : { phase: requestedFilter.phase },
1495
+ ...requestedFilter.tags === void 0 ? {} : { tags: requestedFilter.tags }
1496
+ };
1497
+ const timeoutMs = validateUsageSettlementTimeout(requestedTimeoutMs);
1498
+ const waitResult = ledger.summary(filter).pendingCalls === 0 ? true : ledger.waitForIdle ? await ledger.waitForIdle({ timeoutMs }) : false;
1499
+ const pendingCalls = ledger.summary(filter).pendingCalls;
1500
+ return {
1501
+ settled: waitResult && pendingCalls === 0,
1502
+ pendingCalls,
1503
+ receipt: usageReceiptFromCostLedger(ledger, filter)
1504
+ };
1505
+ }
1506
+ function validateUsageSettlementTimeout(timeoutMs) {
1507
+ const resolved = timeoutMs ?? 5e3;
1508
+ if (!Number.isSafeInteger(resolved) || resolved < 0 || resolved > 2147483647) throw new TypeError("settlementTimeoutMs must be a non-negative safe integer no greater than 2147483647");
1509
+ return resolved;
1510
+ }
1511
+ //#endregion
1512
+ //#region src/analyst/kind-factory.ts
1513
+ /**
1514
+ * Build an `Analyst<TraceAnalysisStore>` from a kind spec.
1515
+ *
1516
+ * Lifts the Ax pipeline once at registration time so the registry
1517
+ * gets a stateless analyst. The Ax agent is freshly constructed per
1518
+ * `analyze()` call (the agent carries chat-log + usage state we don't
1519
+ * want shared across analyst runs).
1520
+ */
1521
+ function createTraceAnalystKind(spec, opts) {
1522
+ rejectRemovedKindOptions(spec);
1523
+ const version = opts.versionSuffix ? `${spec.version}+${opts.versionSuffix}` : spec.version;
1524
+ const model = resolveAnalystModel(opts.ai, opts.model);
1525
+ const minimumEvidenceCitations = spec.minimumEvidenceCitations ?? 1;
1526
+ if (!Number.isInteger(minimumEvidenceCitations) || minimumEvidenceCitations < 1) throw new TypeError("minimumEvidenceCitations must be a positive integer");
1527
+ const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs);
1528
+ return {
1529
+ id: spec.id,
1530
+ description: spec.description,
1531
+ inputKind: "trace-store",
1532
+ cost: {
1533
+ ...spec.cost,
1534
+ settlement_timeout_ms: settlementTimeoutMs
1535
+ },
1536
+ version,
1537
+ async analyze(store, ctx) {
1538
+ const maxOutputTokens = spec.maxOutputTokens ?? 4096;
1539
+ const costLedger = ctx.costLedger ?? new CostLedger(ctx.budgetUsd);
1540
+ const costTags = {
1541
+ ...ctx.tags ?? {},
1542
+ analystId: spec.id,
1543
+ ...ctx.correlationId ? { analystRunId: ctx.correlationId } : {}
1544
+ };
1545
+ const meteredAi = meterAxChatService(opts.ai, {
1546
+ ledger: costLedger,
1547
+ actor: spec.id,
1548
+ maxOutputTokens,
1549
+ defaultModel: model,
1550
+ phase: ctx.costPhase,
1551
+ signal: ctx.signal,
1552
+ tags: costTags
1553
+ });
1554
+ try {
1555
+ const tools = spec.buildTools(store);
1556
+ const maxSubqueries = spec.subqueries?.maxCalls ?? 0;
1557
+ const maxParallel = spec.subqueries?.maxParallel ?? 2;
1558
+ const priorContext = renderPriorFindings(ctx.priorFindings);
1559
+ const upstreamContext = renderUpstreamFindings(ctx.upstreamFindings);
1560
+ const actorDescription = spec.actorDescription.trim() + priorContext + upstreamContext + "\n\n" + RAW_FINDING_SCHEMA_PROMPT + (minimumEvidenceCitations > 1 ? `\n\nThis kind requires at least ${minimumEvidenceCitations} evidence citations per finding; rows with fewer are rejected.` : "") + "\n\nFirst write `report`: a concise free-form prose diagnosis of what the traces show — what succeeded, what was suboptimal or failed — with concrete trace ids and numbers. THEN return the structured `findings` array (it MAY be empty when there is nothing to report).";
1561
+ ctx.log?.(`analyst.kind ${spec.id} forward`, {
1562
+ max_subqueries: maxSubqueries,
1563
+ tool_count: tools.length,
1564
+ tags: ctx.tags
1565
+ });
1566
+ const { report, findings: submittedFindings } = await runTraceAnalysisLoop({
1567
+ id: spec.id,
1568
+ description: spec.description,
1569
+ prompt: actorDescription,
1570
+ question: deriveQuestion(ctx, spec),
1571
+ ai: meteredAi,
1572
+ model,
1573
+ tools,
1574
+ findingType: "object",
1575
+ maxSubqueries,
1576
+ maxParallelSubqueries: maxParallel,
1577
+ maxTurns: spec.maxTurns ?? 12,
1578
+ maxRuntimeChars: spec.maxRuntimeChars ?? 6e3,
1579
+ ...ctx.signal ? { signal: ctx.signal } : {}
1580
+ });
1581
+ const expectedSubjects = KIND_EXPECTED_SUBJECTS[spec.id];
1582
+ const out = [];
1583
+ const rawRows = submittedFindings;
1584
+ let rejectedWrongKind = 0;
1585
+ let rejectedInsufficientEvidence = 0;
1586
+ const processRow = (parsed) => {
1587
+ const callbackResult = spec.postProcess ? spec.postProcess(parsed, ctx) : parsed;
1588
+ if (!callbackResult) return null;
1589
+ const postProcessed = parseRawFinding(callbackResult, ctx.log);
1590
+ if (!postProcessed) return null;
1591
+ if (expectedSubjects && postProcessed.subject !== void 0) {
1592
+ const parsedSubject = parseFindingSubject(postProcessed.subject);
1593
+ if (parsedSubject === null) {
1594
+ ctx.log?.("finding rejected: subject failed to parse", {
1595
+ kind: spec.id,
1596
+ subject: postProcessed.subject
1597
+ });
1598
+ rejectedWrongKind += 1;
1599
+ return null;
1600
+ }
1601
+ if (!expectedSubjects.includes(parsedSubject.kind)) {
1602
+ ctx.log?.("finding rejected: subject variant not allowed for this kind", {
1603
+ kind: spec.id,
1604
+ subject_kind: parsedSubject.kind,
1605
+ subject: postProcessed.subject,
1606
+ allowed: expectedSubjects
1607
+ });
1608
+ rejectedWrongKind += 1;
1609
+ return null;
1610
+ }
1611
+ }
1612
+ const distinctEvidenceCitations = new Set(postProcessed.evidence.map((citation) => citation.uri.trim())).size;
1613
+ if (distinctEvidenceCitations < minimumEvidenceCitations) {
1614
+ ctx.log?.("finding rejected: insufficient evidence citations", {
1615
+ kind: spec.id,
1616
+ required: minimumEvidenceCitations,
1617
+ received: postProcessed.evidence.length,
1618
+ distinct: distinctEvidenceCitations
1619
+ });
1620
+ rejectedInsufficientEvidence += 1;
1621
+ return null;
1622
+ }
1623
+ return postProcessed;
1624
+ };
1625
+ for (const row of rawRows) {
1626
+ const parsed = parseRawFinding(row, ctx.log);
1627
+ if (!parsed) continue;
1628
+ const postProcessed = processRow(parsed);
1629
+ if (!postProcessed) continue;
1630
+ out.push(toAnalystFinding(spec, version, postProcessed));
1631
+ }
1632
+ ctx.log?.(`analyst.kind ${spec.id} done`, {
1633
+ emitted: rawRows.length,
1634
+ accepted: out.length,
1635
+ rejected_wrong_subject: rejectedWrongKind,
1636
+ rejected_insufficient_evidence: rejectedInsufficientEvidence
1637
+ });
1638
+ if (out.length === 0 && report.trim().length >= 200) {
1639
+ if (opts.recovery) {
1640
+ const wrongKindBefore = rejectedWrongKind;
1641
+ const insufficientEvidenceBefore = rejectedInsufficientEvidence;
1642
+ const recovered = await structureFindings({
1643
+ report,
1644
+ analystId: spec.id,
1645
+ area: spec.area,
1646
+ model: opts.recovery.model ?? model,
1647
+ baseUrl: opts.recovery.baseUrl,
1648
+ apiKey: opts.recovery.apiKey,
1649
+ fetchImpl: opts.recovery.fetchImpl,
1650
+ costLedger,
1651
+ costPhase: ctx.costPhase,
1652
+ costTags,
1653
+ signal: ctx.signal,
1654
+ maxTokens: Math.min(maxOutputTokens, 2e3),
1655
+ processRow,
1656
+ findingMetadata: { kind_version: version }
1657
+ });
1658
+ out.push(...recovered.findings);
1659
+ ctx.log?.(`analyst.kind ${spec.id} recovery`, {
1660
+ outcome: recovered.outcome,
1661
+ recovered: recovered.findings.length,
1662
+ rejected_wrong_subject: rejectedWrongKind - wrongKindBefore,
1663
+ rejected_insufficient_evidence: rejectedInsufficientEvidence - insufficientEvidenceBefore
1664
+ });
1665
+ }
1666
+ if (out.length === 0) {
1667
+ const fallback = processRow({
1668
+ claim: "Analyst produced a diagnosis but no structured findings — see report.",
1669
+ rationale: report.slice(0, 1500),
1670
+ severity: "info",
1671
+ confidence: .3,
1672
+ evidence: [{
1673
+ uri: "report://summary",
1674
+ excerpt: report.slice(0, 2e3)
1675
+ }]
1676
+ });
1677
+ if (fallback) out.push(toAnalystFinding(spec, version, fallback, { outcome: "extraction_failed" }));
1678
+ else throw new Error(`Trace analyst '${spec.id}' produced a substantive report, but no finding satisfied its acceptance rules`);
1679
+ }
1680
+ }
1681
+ return out;
1682
+ } finally {
1683
+ const usage = await settleUsageReceiptFromCostLedger(costLedger, {
1684
+ tags: {
1685
+ analystId: spec.id,
1686
+ ...ctx.correlationId ? { analystRunId: ctx.correlationId } : {}
1687
+ },
1688
+ timeoutMs: settlementTimeoutMs
1689
+ });
1690
+ if (!usage.settled) ctx.log?.(`analyst.kind ${spec.id} provider settlement timed out`, {
1691
+ pending_calls: usage.pendingCalls,
1692
+ timeout_ms: settlementTimeoutMs
1693
+ });
1694
+ ctx.recordUsage?.(usage.receipt);
1695
+ }
1696
+ }
1697
+ };
1698
+ }
1699
+ function rejectRemovedKindOptions(spec) {
1700
+ const supplied = spec;
1701
+ for (const [removed, replacement] of [
1702
+ ["recursion", "subqueries"],
1703
+ ["responderDescription", "actorDescription"],
1704
+ ["maxDepth", "subqueries"],
1705
+ ["maxParallelSubagents", "subqueries.maxParallel"],
1706
+ ["subagentDescription", "actorDescription"]
1707
+ ]) if (removed in supplied) throw new TypeError(`createTraceAnalystKind: '${removed}' is unsupported; use '${replacement}'`);
1708
+ }
1709
+ function deriveQuestion(ctx, spec) {
1710
+ const focus = ctx.tags?.focus?.trim();
1711
+ const task = `Analyze this trace dataset with the available tools and report ${spec.area} findings. ${spec.description}`;
1712
+ return focus ? `${task} Focus: ${focus}.` : task;
1713
+ }
1714
+ function toAnalystFinding(spec, version, raw, metadata = {}) {
1715
+ return makeFinding({
1716
+ analyst_id: spec.id,
1717
+ area: spec.area,
1718
+ subject: raw.subject,
1719
+ claim: raw.claim,
1720
+ rationale: raw.rationale,
1721
+ severity: raw.severity,
1722
+ confidence: raw.confidence,
1723
+ evidence_refs: evidenceRefsFromRawFinding(raw),
1724
+ recommended_action: raw.recommended_action,
1725
+ metadata: {
1726
+ kind_version: version,
1727
+ ...metadata
1728
+ }
1729
+ });
1730
+ }
1731
+ /**
1732
+ * Render a compact prior-findings block the actor reads alongside its
1733
+ * brief. Each row is one line so the actor can scan dozens cheaply.
1734
+ * The kind's prompt instructs the actor to (a) check whether a new
1735
+ * cluster matches a prior `finding_id` (carry the id forward via
1736
+ * `id_basis` to keep diffs stable) and (b) raise severity / confidence
1737
+ * when a prior finding has reappeared without remediation.
1738
+ *
1739
+ * Returns the empty string when there are no prior findings — most
1740
+ * runs are "first-of-its-kind" and the prompt stays unchanged.
1741
+ *
1742
+ * Exported for tests + for consumers that build their own actor
1743
+ * prompts (e.g. specialized analysts living outside the default kinds).
1744
+ */
1745
+ function renderPriorFindings(prior) {
1746
+ if (!prior || prior.length === 0) return "";
1747
+ const MAX_ROWS = 40;
1748
+ const rows = prior.slice(0, MAX_ROWS).map((f) => {
1749
+ const subject = f.subject ? ` [${f.subject}]` : "";
1750
+ return ` - id=${f.finding_id} ${f.severity}${subject} ${truncateForContext(f.claim, 160)}`;
1751
+ });
1752
+ const overflow = prior.length > MAX_ROWS ? `\n ... +${prior.length - MAX_ROWS} more prior findings (older history truncated)` : "";
1753
+ return [
1754
+ "",
1755
+ "",
1756
+ "PRIOR FINDINGS (from a previous run on related data):",
1757
+ "When the work you do now matches a row below, REUSE the `finding_id` (pass it as `id_basis`) so the cross-run diff stays stable.",
1758
+ "A finding that reappears with no remediation evidence SHOULD raise its `confidence` and may justify a higher `severity`.",
1759
+ ...rows,
1760
+ overflow
1761
+ ].filter(Boolean).join("\n");
1762
+ }
1763
+ /** Render findings produced earlier in this same registry run. */
1764
+ function renderUpstreamFindings(upstream) {
1765
+ if (!upstream || upstream.length === 0) return "";
1766
+ const MAX_ROWS = 40;
1767
+ const rows = upstream.slice(0, MAX_ROWS).map((finding) => {
1768
+ const subject = finding.subject ? ` [${finding.subject}]` : "";
1769
+ const action = finding.recommended_action ? ` action=${truncateForContext(finding.recommended_action, 120)}` : "";
1770
+ const evidence = finding.evidence_refs[0] ? ` evidence=${truncateForContext(finding.evidence_refs[0].uri, 120)}` : "";
1771
+ return ` - id=${finding.finding_id} source=${finding.analyst_id} ${finding.severity}${subject} claim=${truncateForContext(finding.claim, 160)}${action}${evidence}`;
1772
+ });
1773
+ const overflow = upstream.length > MAX_ROWS ? `\n ... +${upstream.length - MAX_ROWS} more upstream findings (truncated)` : "";
1774
+ return [
1775
+ "",
1776
+ "",
1777
+ "UPSTREAM FINDINGS (produced earlier in this same registry run):",
1778
+ "Use these as intermediate evidence. Build on them instead of repeating the same diagnosis, and cite a dependency with `finding://<id>`.",
1779
+ ...rows,
1780
+ overflow
1781
+ ].filter(Boolean).join("\n");
1782
+ }
1783
+ function truncateForContext(s, max) {
1784
+ if (s.length <= max) return s;
1785
+ return `${s.slice(0, max - 1).trimEnd()}…`;
1786
+ }
1787
+ //#endregion
1788
+ //#region src/analyst/tool-groups.ts
1789
+ const TOOL_NAMES_BY_GROUP = {
1790
+ all: /* @__PURE__ */ new Set(),
1791
+ discovery: /* @__PURE__ */ new Set([
1792
+ "getDatasetOverview",
1793
+ "queryTraces",
1794
+ "countTraces"
1795
+ ]),
1796
+ discoveryAndRead: /* @__PURE__ */ new Set([
1797
+ "getDatasetOverview",
1798
+ "queryTraces",
1799
+ "countTraces",
1800
+ "viewTrace",
1801
+ "viewSpans"
1802
+ ]),
1803
+ discoveryAndSearch: /* @__PURE__ */ new Set([
1804
+ "getDatasetOverview",
1805
+ "queryTraces",
1806
+ "countTraces",
1807
+ "searchTrace",
1808
+ "searchSpan"
1809
+ ]),
1810
+ targeted: /* @__PURE__ */ new Set([
1811
+ "getDatasetOverview",
1812
+ "queryTraces",
1813
+ "viewSpans",
1814
+ "searchSpan"
1815
+ ])
1816
+ };
1817
+ /**
1818
+ * Build the tool set for a named group bound to a specific trace store.
1819
+ *
1820
+ * `all` returns every tool. Other groups filter `buildTraceAnalystTools`
1821
+ * by name to the documented subset. An unrecognised group name throws —
1822
+ * silently returning all tools would defeat the cost-control point.
1823
+ */
1824
+ function buildTraceToolsForGroup(group, store) {
1825
+ const all = buildTraceAnalystTools({ store });
1826
+ if (group === "all") return all;
1827
+ const allow = TOOL_NAMES_BY_GROUP[group];
1828
+ if (!allow) throw new Error(`unknown trace tool group: ${group}`);
1829
+ return all.filter((tool) => allow.has(tool.name));
1830
+ }
1831
+ const FAILURE_MODE_KIND_SPEC = {
1832
+ id: "failure-mode",
1833
+ description: "Clusters trace-dataset failures into distinct failure modes with cited evidence and a short recommended action.",
1834
+ area: "failure-mode",
1835
+ version: "1.2.0",
1836
+ actorDescription: `You are a failure-mode classifier for an OTLP trace dataset. Your job is to identify the **distinct ways agents failed** in this dataset, not to grade individual runs.
1837
+
1838
+ ${findingSubjectGrammarPromptFor("failure-mode")}
1839
+
1840
+ DISCOVERY → CLUSTER → CITE protocol:
1841
+
1842
+ 1. Call \`traces.getDatasetOverview({})\` first. Use \`has_errors\`, \`models\`, \`agent_names\`, \`tools\`, and \`sample_trace_ids\` to size the failure surface.
1843
+ 2. Use \`traces.queryTraces({ filters: { has_errors: true }, limit })\` to pull error-bearing traces. Combine with \`traces.countTraces\` to see what fraction of the dataset failed.
1844
+ 3. For each candidate failure cluster, use \`traces.searchTrace\` with regex like \`STATUS_CODE_ERROR\`, \`MaxTurnsExceeded\`, \`assertion\`, \`unauthorized\`, \`timeout\`, \`429\`, \`5\\d\\d\`, the agent's specific error strings, or the names of its tools. Pull one or two representative traces per cluster, **not all** of them.
1845
+ 4. **Cluster, do not enumerate.** Two failures with the same root cause should be ONE finding citing both traces, not two findings. The point of this analyst is to compress N runs into K modes.
1846
+ 5. For each defensible cluster, emit ONE finding. Use a lowercase cluster label matching the subject grammar ("tool-call-loop", "auth-revoked-mid-run", ...). Rate it critical when it blocks the run, high when the run finishes degraded, and medium when it slows convergence. Cite representative spans and include exact error, payload, or contradictory-output quotes. Use confidence 0.85+ when multiple traces show the same shape, 0.6-0.8 for a single-trace inference, and <0.5 for speculation. Keep the imperative fix idea short; the improvement analyst expands it.
1847
+
1848
+ If the dataset has no failures, return an empty findings array — do NOT pad with low-confidence speculation.
1849
+
1850
+ **Use subqueries over loaded evidence.** After the first scan, load representative span excerpts for each candidate cluster. Then send one bounded \`llmQuery\` per cluster in one batch, including the exact excerpts and asking it to classify the root cause. Subqueries cannot call trace tools. Merge or split clusters yourself from their classifications and the cited source evidence.
1851
+
1852
+ OBSERVABILITY rules:
1853
+ - Each non-final turn must emit at least one \`console.log\` for evidence.
1854
+ - Reuse runtime variables across turns; don't recompute.`,
1855
+ buildTools: (store) => buildTraceToolsForGroup("all", store),
1856
+ subqueries: {
1857
+ maxCalls: 8,
1858
+ maxParallel: 4
1859
+ },
1860
+ maxTurns: 24,
1861
+ cost: { kind: "llm" }
1862
+ };
1863
+ const IMPROVEMENT_KIND_SPEC = {
1864
+ id: "improvement",
1865
+ description: "Converts upstream failure / gap / poisoning findings into concrete locus-named edits (prompt, tool-doc, RAG, scaffolding) with leverage grades.",
1866
+ area: "improvement",
1867
+ version: "1.2.0",
1868
+ actorDescription: `You are a self-improvement analyst. Your job is to propose **concrete, locus-named edits** the agent's runtime should adopt to fix the failure modes, knowledge gaps, and poisonings present in this dataset.
1869
+
1870
+ Upstream analysts have already classified the problems. Your job is to convert each problem into a *change to make* and grade its expected leverage. Each finding is one proposed edit.
1871
+
1872
+ ${findingSubjectGrammarPromptFor("improvement")}
1873
+
1874
+ DISCOVERY → CANDIDATE-FIXES → COMPETE → CITE protocol:
1875
+
1876
+ 1. \`traces.getDatasetOverview({})\` first. Note the agents, tools, and any system-prompt fingerprints (look for the prompt text echoed in early spans).
1877
+ 2. For each high-severity failure pattern, generate 2-3 candidate fixes. Real candidate axes:
1878
+ - **System-prompt edit** — add an instruction, remove a misleading one, restructure precedence
1879
+ - **Tool description edit** — rewrite a tool's description so the agent picks it correctly / passes valid args
1880
+ - **New tool** — add a tool the agent kept emulating in code
1881
+ - **RAG ingestion** — add a document or correct a stale one
1882
+ - **Memory invalidation** — clear cached prior-run decisions that no longer apply
1883
+ - **Scaffolding** — add a precondition check, a retry policy, a turn budget, a verification step
1884
+ - **Output schema** — narrow the agent's output to forbid the failure shape
1885
+ - **Skill / MCP / hook / subagent** — change the reusable profile component responsible for the behavior
1886
+ - **Workflow / rollout policy** — change orchestration, budget, sampling, or stopping behavior
1887
+ - **Code** — change an implementation path when profile edits cannot repair the behavior
1888
+ 3. **Compare candidate fixes with bounded subqueries.** Load the representative failure excerpts, then send one \`llmQuery\` per candidate-fix axis the same evidence. Ask for likely effect, side effects, and implementation scope. Subqueries cannot call trace tools; trace ids alone are insufficient context.
1889
+ 4. After the comparisons return, **pick the winning candidate per cluster** based on expected effect and risk, then emit ONE finding. Keep the alternatives and rejection reasons in the rationale so the recommendation is auditable.
1890
+ 5. **Cross-reference upstream findings.** Cite prior failure-mode or knowledge-gap findings as \`finding://<prior-finding-id>\`. This builds the dependency graph that lets the dashboard show "fix #X resolves failure modes A, B, C."
1891
+
1892
+ For each winning recommendation, emit ONE finding. Use one exact locus from the subject grammar and state the edit in one sentence. Match leverage to the source failure's severity; use medium for quality-of-life changes and info for cleanup with no behavioral effect. Cite the targeted \`finding://<id>\` when available and the most representative span when useful. Quote the problem being fixed. Use confidence 0.85+ for a mechanical fix to a well-evidenced failure, 0.6-0.8 when judgment is required, and <0.5 for speculation. Explain in at most two sentences why this candidate beat its alternatives. The recommended action must be the literal diff, quoted replacement, tool description, or setting change.
1893
+
1894
+ If no upstream failure findings exist in this run, derive your own from the trace dataset using the failure-mode protocol inline (\`searchTrace\` for STATUS_CODE_ERROR / MaxTurnsExceeded / etc.). But prefer to consume upstream findings when present — the kinds are designed to chain.
1895
+
1896
+ Do NOT propose a fix you cannot defend with evidence. "Tighten the prompt" is not a finding; "Add 'When the user asks for X, always Y' to the system prompt section "request-classification"" is.
1897
+
1898
+ OBSERVABILITY rules:
1899
+ - Each non-final turn must emit at least one \`console.log\` for evidence.`,
1900
+ buildTools: (store) => buildTraceToolsForGroup("all", store),
1901
+ subqueries: {
1902
+ maxCalls: 8,
1903
+ maxParallel: 4
1904
+ },
1905
+ maxTurns: 30,
1906
+ maxRuntimeChars: 12e3,
1907
+ cost: { kind: "llm" }
1908
+ };
1909
+ const KNOWLEDGE_GAP_KIND_SPEC = {
1910
+ id: "knowledge-gap",
1911
+ description: "Identifies missing or stale pieces of knowledge — primarily against the agent-knowledge wiki — and attributes each to the runtime layer (wiki page, claim, raw source, websearch, tool-doc, system-prompt, memory) that should have held it.",
1912
+ area: "knowledge-gap",
1913
+ version: "1.2.0",
1914
+ actorDescription: `You are a knowledge-gap analyst for an OTLP trace dataset. Your job is to identify the **specific pieces of information the agent lacked, or that were stale**, that caused poor decisions.
1915
+
1916
+ The agent under analysis maintains a curated knowledge base via \`@tangle-network/agent-knowledge\` — a wiki of \`KnowledgePage\`s with raw source anchors, claims, and relations. The primary expected store of agent-knowable facts IS that wiki. A "knowledge gap" is anything the agent had to discover or guess at run-time that the wiki should have held — or an outdated/contradictory fact the agent picked up from a non-wiki source.
1917
+
1918
+ ${findingSubjectGrammarPromptFor("knowledge-gap")}
1919
+
1920
+ DISCOVERY → ATTRIBUTE-TO-LAYER → CITE protocol:
1921
+
1922
+ 1. \`traces.getDatasetOverview({})\` first. Note which agents, tools, and models appear.
1923
+ 2. Pull traces where the agent shows gap signals. The strongest signals are:
1924
+ - Self-correction turns ("I assumed X but…", "let me re-check", "actually,")
1925
+ - Clarifying-question turns where the agent asked the user something the runtime should have surfaced
1926
+ - Repeated retrieval / lookup calls for the same artifact with slightly varied queries
1927
+ - Tool errors that name a missing argument or unknown resource
1928
+ - Web-search calls returning pages dated before a known cutoff for content that changes (versioned APIs, schemas, policies)
1929
+ - Agent quoting a tool's docs / system prompt incorrectly because the actual text was insufficient
1930
+ - Fabricated identifiers that don't appear in dataset \`sample_trace_ids\`
1931
+ Use \`traces.searchTrace\` with patterns like \`I (don.?t|do not) know\`, \`assumed\`, \`unclear\`, \`could you (clarify|tell me|provide)\`, \`not found\`, \`undefined\`, \`unknown\`, \`null\`, dates older than the analysis window, or the agent's specific clarification phrases.
1932
+ 3. For each gap, identify the **layer of the runtime that should have prevented it** and use its exact locus from the subject grammar above.
1933
+ 4. For each defensible gap, emit ONE finding. Use an exact locus from the subject grammar and name the missing or stale knowledge (for example, "wiki has no page on invoice line-item shape; agent re-derived it from raw spans"). Rate it high when it caused failure or a clarifying question, medium for unnecessary turns, and low for minor inefficiency. Cite the span where the question, correction, retrieval miss, or stale result surfaced and quote it exactly. Use confidence 0.85+ when the agent articulated the gap and 0.6-0.8 when inferred. Recommend a concrete wiki edit for an agent-knowledge locus or a prompt/tool-description edit otherwise.
1934
+
1935
+ **Compare layers over loaded evidence.** After the first scan, load the exact excerpts behind candidates across \`agent-knowledge:*\`, \`websearch:outdated\`, \`tool-doc:*\`, \`system-prompt:*\`, and \`memory:*\`. Use one bounded \`llmQuery\` per layer to classify those excerpts. Subqueries cannot call trace tools. Merge their classifications into the final finding set only when the source excerpts support them.
1936
+
1937
+ Do NOT report a gap that the agent later recovered from cleanly within the same turn — that's resilience, not a gap. Cite the *non-recovery* version when both exist.
1938
+
1939
+ OBSERVABILITY rules:
1940
+ - Each non-final turn must emit at least one \`console.log\` for evidence.`,
1941
+ buildTools: (store) => buildTraceToolsForGroup("discoveryAndSearch", store),
1942
+ subqueries: {
1943
+ maxCalls: 5,
1944
+ maxParallel: 4
1945
+ },
1946
+ maxTurns: 18,
1947
+ cost: { kind: "llm" }
1948
+ };
1949
+ const KNOWLEDGE_POISONING_KIND_SPEC = {
1950
+ id: "knowledge-poisoning",
1951
+ description: "Identifies confident-but-wrong actions caused by stale memory, contradicting RAG, deprecated tool docs, or outdated system-prompt instructions.",
1952
+ area: "knowledge-poisoning",
1953
+ version: "1.2.0",
1954
+ actorDescription: `You are a knowledge-poisoning analyst for an OTLP trace dataset. Your job is to identify cases where the agent **confidently used wrong information** — not where it lacked information (that's the knowledge-gap analyst).
1955
+
1956
+ ${findingSubjectGrammarPromptFor("knowledge-poisoning")}
1957
+
1958
+ DISCOVERY → DUAL-VERIFY → CITE protocol:
1959
+
1960
+ 1. \`traces.getDatasetOverview({})\` first. Identify the agents, models, and tools.
1961
+ 2. Pull traces where the agent's confident action was later contradicted. Strongest signals:
1962
+ - Agent stated a fact in one span; a later span surfaced contradictory evidence; the agent then proceeded anyway or fabricated reconciliation.
1963
+ - Tool call with stale arguments (an id that no longer exists, an API shape that changed).
1964
+ - Agent cited an \`agent-knowledge\` wiki page or claim whose content contradicts the trace's own evidence — the wiki itself drifted.
1965
+ - Web-search result the agent cited that returned an outdated page; agent treated it as canonical.
1966
+ - System-prompt instruction the agent followed that ground-truth evidence in the trace contradicts (e.g. prompt says "use endpoint A"; tool reply says "endpoint A deprecated, use B").
1967
+ - Repeated wrong-shape parsing despite the tool's actual output proving the shape.
1968
+ 3. Use \`traces.searchTrace\` with regex on phrases like \`actually\`, \`turns out\`, \`previously assumed\`, \`old version\`, \`deprecated\`, \`updated to\`, \`now uses\`, or specific entity names you suspect have changed.
1969
+ 4. For each candidate poisoning, **DUAL-VERIFY**:
1970
+ - Confirm the agent actually acted on the false belief (cite the span where it did)
1971
+ - Confirm the belief is actually false in this trace's own evidence (cite the span that contradicts it)
1972
+ Only emit a finding when both halves are nailed down. If you can only nail one, drop it — single-evidence poisoning findings are too speculative to be useful.
1973
+
1974
+ **Independently assess both halves.** Load the action excerpt and contradicting excerpt yourself, then send bounded \`llmQuery\` calls the exact evidence for "did the agent act?" and "does the trace contradict the belief?" Subqueries cannot call trace tools. Accept a poisoning only when both assessments and the source excerpts support it.
1975
+
1976
+ For each confirmed poisoning, emit ONE finding. Use the source of the false belief as the exact subject. State "agent believed X (from source S); trace evidence shows X is false." Rate it critical for a wrong user-visible action, high when caught internally after significant waste, and medium for inefficiency. Cite BOTH the action span and the contradicting span with exact quotes. Use confidence 0.85+ when both halves have exact quotes and 0.6-0.8 when one half is inferred. Recommend the literal source correction: update the wiki claim, invalidate and re-curate the raw source, or replace the stale prompt/tool instruction.
1977
+
1978
+ Do NOT report a finding if the agent caught and corrected the false belief in the same turn — that's the system working. Reserve poisoning for cases where the false belief shaped downstream action.
1979
+
1980
+ OBSERVABILITY rules:
1981
+ - Each non-final turn must emit at least one \`console.log\` for evidence.`,
1982
+ buildTools: (store) => buildTraceToolsForGroup("all", store),
1983
+ subqueries: {
1984
+ maxCalls: 8,
1985
+ maxParallel: 4
1986
+ },
1987
+ maxTurns: 20,
1988
+ minimumEvidenceCitations: 2,
1989
+ cost: { kind: "llm" }
1990
+ };
1991
+ //#endregion
1992
+ //#region src/analyst/kinds/index.ts
1993
+ /**
1994
+ * The default kind suite. Order is the run order operators should
1995
+ * use: failure-mode first (no upstream deps), gap + poisoning next
1996
+ * (both depend on failures), improvement last (chains all three).
1997
+ */
1998
+ const DEFAULT_TRACE_ANALYST_KINDS = [
1999
+ FAILURE_MODE_KIND_SPEC,
2000
+ KNOWLEDGE_GAP_KIND_SPEC,
2001
+ KNOWLEDGE_POISONING_KIND_SPEC,
2002
+ IMPROVEMENT_KIND_SPEC
2003
+ ];
2004
+ //#endregion
2005
+ //#region src/analyst/registry.ts
2006
+ /**
2007
+ * AnalystRegistry — orchestrate N analysts against one run.
2008
+ *
2009
+ * Owns three responsibilities and only three:
2010
+ * 1. Registration — ids must be unique; bad registrations fail loudly
2011
+ * at register-time, not run-time.
2012
+ * 2. Routing — each analyst declares its `inputKind`; the registry
2013
+ * picks the matching field from AnalystRunInputs and skips the
2014
+ * analyst with a logged reason if it's missing.
2015
+ * 3. Isolation — one analyst's exception MUST NOT stop other analysts.
2016
+ * Failed analysts produce zero findings + a 'failed' summary row.
2017
+ *
2018
+ * Cross-cutting concerns (telemetry, error → finding conversion, cost
2019
+ * ingestion, storage rotation) live in `AnalystHooks`. Budget shaping
2020
+ * (equal split vs weighted vs custom) lives in `BudgetPolicy`. Both
2021
+ * have sensible defaults; consumers override only what they need.
2022
+ */
2023
+ var AnalystRegistry = class {
2024
+ analysts = /* @__PURE__ */ new Map();
2025
+ options;
2026
+ constructor(options = {}) {
2027
+ this.options = options;
2028
+ }
2029
+ register(analyst) {
2030
+ if (!analyst.id) throw new Error("AnalystRegistry.register: analyst.id is required");
2031
+ if (this.analysts.has(analyst.id)) throw new Error(`AnalystRegistry.register: duplicate analyst id "${analyst.id}"`);
2032
+ if (!analyst.version) throw new Error(`AnalystRegistry.register: analyst "${analyst.id}" must declare a version`);
2033
+ if (analyst.cost.kind === "deterministic" && analyst.cost.settlement_timeout_ms !== void 0) throw new TypeError(`AnalystRegistry.register: deterministic analyst "${analyst.id}" cannot declare settlement_timeout_ms`);
2034
+ if (analyst.cost.settlement_timeout_ms !== void 0) validateUsageSettlementTimeout(analyst.cost.settlement_timeout_ms);
2035
+ this.analysts.set(analyst.id, analyst);
2036
+ }
2037
+ list() {
2038
+ return Array.from(this.analysts.values()).map((a) => ({
2039
+ id: a.id,
2040
+ description: a.description,
2041
+ version: a.version,
2042
+ cost: a.cost
2043
+ }));
2044
+ }
2045
+ async run(runId, inputs, runOpts = {}) {
2046
+ for await (const ev of this.runStream(runId, inputs, runOpts)) if (ev.type === "run-completed") return ev.result;
2047
+ throw new Error("AnalystRegistry.run: stream completed without run-completed event");
2048
+ }
2049
+ /**
2050
+ * Streaming counterpart to `run()`. Emits `AnalystRunEvent` values
2051
+ * in real time — `run-started`, then per-analyst `skipped` /
2052
+ * `started` / `completed`, then a terminal `run-completed` whose
2053
+ * payload is the full `AnalystRunResult`. UIs use this to render
2054
+ * progress; persistence consumers use `run()` and read the result.
2055
+ *
2056
+ * Hooks (`onBeforeAnalyze` / `onAfterAnalyze` / `onError` /
2057
+ * `onComplete`) fire as before — streaming is additive, not a hook
2058
+ * replacement.
2059
+ */
2060
+ async *runStream(runId, inputs, runOpts = {}) {
2061
+ const correlationId = `ar_${randomUUID().slice(0, 12)}`;
2062
+ const log = this.options.log ?? (() => {});
2063
+ const hooks = this.options.hooks ?? {};
2064
+ const startedAt = (/* @__PURE__ */ new Date()).toISOString();
2065
+ const started = Date.now();
2066
+ const timeoutMs = validateTimeout(runOpts.timeoutMs);
2067
+ const deadlineMs = timeoutMs === void 0 ? void 0 : started + timeoutMs;
2068
+ const timeoutSignal = timeoutMs === void 0 ? void 0 : AbortSignal.timeout(timeoutMs);
2069
+ const runSignal = combineAbortSignals(runOpts.signal, timeoutSignal);
2070
+ const selected = this.selectAnalysts(runOpts);
2071
+ const budget = runOpts.budget ?? this.options.defaultBudget;
2072
+ validateBudgetPolicy(budget);
2073
+ yield {
2074
+ type: "run-started",
2075
+ run_id: runId,
2076
+ correlation_id: correlationId,
2077
+ started_at: startedAt,
2078
+ analyst_ids: selected.map((a) => a.id)
2079
+ };
2080
+ const summaries = [];
2081
+ const allFindings = [];
2082
+ let totalCost = 0;
2083
+ let remainingUsd = budget?.totalUsd;
2084
+ const runnableAnalysts = selected.filter((a) => this.routeInput(a, inputs).kind !== "missing");
2085
+ const runnableCount = runnableAnalysts.length;
2086
+ const weights = budget?.weights;
2087
+ const totalWeight = weights && budget?.totalUsd != null && !budget.allocate && runnableCount > 0 ? runnableAnalysts.reduce((sum, analyst) => sum + analystWeight(weights, analyst.id), 0) : void 0;
2088
+ if (totalWeight === 0) throw new Error("BudgetPolicy.weights must allocate positive weight to a runnable analyst");
2089
+ for (const analyst of selected) {
2090
+ const t0 = Date.now();
2091
+ if (runSignal?.aborted) {
2092
+ const summary = abortedBeforeStartSummary(analyst, runSignal);
2093
+ summaries.push(summary);
2094
+ log(`[analyst] skip ${analyst.id} — run aborted`, {
2095
+ runId,
2096
+ reason: summary.reason
2097
+ });
2098
+ yield {
2099
+ type: "analyst-skipped",
2100
+ summary
2101
+ };
2102
+ continue;
2103
+ }
2104
+ const input = this.routeInput(analyst, inputs);
2105
+ if (input.kind === "missing") {
2106
+ const summary = {
2107
+ analyst_id: analyst.id,
2108
+ status: "skipped",
2109
+ reason: `missing input of kind '${analyst.inputKind}'`,
2110
+ findings_count: 0,
2111
+ latency_ms: 0,
2112
+ usage: zeroUsage()
2113
+ };
2114
+ summaries.push(summary);
2115
+ log(`[analyst] skip ${analyst.id} — missing input`, {
2116
+ runId,
2117
+ kind: analyst.inputKind
2118
+ });
2119
+ await waitForHook(hooks.onAfterAnalyze ? () => hooks.onAfterAnalyze?.({
2120
+ analyst,
2121
+ summary,
2122
+ findings: [],
2123
+ runId
2124
+ }) : void 0, runSignal);
2125
+ yield {
2126
+ type: "analyst-skipped",
2127
+ summary
2128
+ };
2129
+ continue;
2130
+ }
2131
+ const perBudget = allocateBudget(budget, {
2132
+ analyst,
2133
+ remainingUsd,
2134
+ runningCount: runnableCount,
2135
+ totalWeight
2136
+ });
2137
+ const usageReceipts = [];
2138
+ const ctx = {
2139
+ runId,
2140
+ correlationId,
2141
+ deadlineMs,
2142
+ budgetUsd: perBudget,
2143
+ costLedger: runOpts.costLedger,
2144
+ costPhase: runOpts.costPhase,
2145
+ chat: this.options.chat,
2146
+ tags: runOpts.tags,
2147
+ log: (msg, fields) => log(`[${analyst.id}] ${msg}`, {
2148
+ runId,
2149
+ correlationId,
2150
+ ...fields
2151
+ }),
2152
+ signal: runSignal,
2153
+ priorFindings: selectPriorFindings(runOpts.priorFindings, analyst.id),
2154
+ upstreamFindings: runOpts.chainFindings && allFindings.length > 0 ? [...allFindings] : void 0,
2155
+ recordUsage: (receipt) => {
2156
+ assertValidUsageReceipt(receipt);
2157
+ usageReceipts.push(receipt);
2158
+ }
2159
+ };
2160
+ await waitForHook(hooks.onBeforeAnalyze ? () => hooks.onBeforeAnalyze?.({
2161
+ analyst,
2162
+ ctx,
2163
+ runId
2164
+ }) : void 0, runSignal);
2165
+ if (runSignal?.aborted) {
2166
+ const summary = abortedBeforeStartSummary(analyst, runSignal, Date.now() - t0);
2167
+ summaries.push(summary);
2168
+ log(`[analyst] skip ${analyst.id} — run aborted`, {
2169
+ runId,
2170
+ reason: summary.reason
2171
+ });
2172
+ yield {
2173
+ type: "analyst-skipped",
2174
+ summary
2175
+ };
2176
+ continue;
2177
+ }
2178
+ const effectiveBudget = validateEffectiveBudget(ctx.budgetUsd, remainingUsd, analyst.id);
2179
+ yield {
2180
+ type: "analyst-started",
2181
+ analyst_id: analyst.id,
2182
+ started_at: new Date(t0).toISOString()
2183
+ };
2184
+ let findings;
2185
+ let summary;
2186
+ try {
2187
+ if (runSignal?.aborted) throw abortReason(runSignal);
2188
+ findings = await waitForOperation(analyst.analyze(input.value, ctx), runSignal, analystAbortGraceMs(analyst));
2189
+ const latency = Date.now() - t0;
2190
+ const usage = resolveUsage(analyst, usageReceipts);
2191
+ const cost = knownCostUsd(usage);
2192
+ totalCost += cost;
2193
+ if (typeof remainingUsd === "number") remainingUsd = Math.max(0, remainingUsd - budgetDebit(usage, effectiveBudget));
2194
+ allFindings.push(...findings);
2195
+ summary = {
2196
+ analyst_id: analyst.id,
2197
+ status: "ok",
2198
+ findings_count: findings.length,
2199
+ latency_ms: latency,
2200
+ usage
2201
+ };
2202
+ summaries.push(summary);
2203
+ log(`[analyst] ok ${analyst.id}`, {
2204
+ runId,
2205
+ findings: findings.length,
2206
+ latency_ms: latency,
2207
+ cost_usd: cost,
2208
+ cost_kind: usage.cost.kind,
2209
+ input_tokens: usage.tokens?.input ?? null,
2210
+ output_tokens: usage.tokens?.output ?? null
2211
+ });
2212
+ if (effectiveBudget !== void 0 && usage.cost.kind === "uncaptured") log(`[analyst] WARN ${analyst.id} — USD cost uncaptured; budget not reconciled`, {
2213
+ runId,
2214
+ budget_usd: effectiveBudget,
2215
+ cost_captured: false
2216
+ });
2217
+ } catch (err) {
2218
+ const latency = Date.now() - t0;
2219
+ const e = err instanceof Error ? err : new Error(String(err));
2220
+ const hookFindings = runSignal?.aborted ? [] : await hooks.onError?.({
2221
+ analyst,
2222
+ error: e,
2223
+ runId
2224
+ }) ?? [];
2225
+ if (hookFindings.length) allFindings.push(...hookFindings);
2226
+ const usage = resolveUsage(analyst, usageReceipts);
2227
+ const cost = knownCostUsd(usage);
2228
+ totalCost += cost;
2229
+ if (typeof remainingUsd === "number") remainingUsd = Math.max(0, remainingUsd - budgetDebit(usage, effectiveBudget));
2230
+ const summary = {
2231
+ analyst_id: analyst.id,
2232
+ status: "failed",
2233
+ findings_count: hookFindings.length,
2234
+ latency_ms: latency,
2235
+ usage,
2236
+ error: {
2237
+ class: e.constructor.name,
2238
+ message: e.message
2239
+ }
2240
+ };
2241
+ summaries.push(summary);
2242
+ log(`[analyst] FAIL ${analyst.id}`, {
2243
+ runId,
2244
+ error_class: e.constructor.name,
2245
+ error: e.message,
2246
+ cost_usd: cost,
2247
+ cost_kind: usage.cost.kind
2248
+ });
2249
+ if (effectiveBudget !== void 0 && usage.cost.kind === "uncaptured") log(`[analyst] WARN ${analyst.id} — USD cost uncaptured; budget not reconciled`, {
2250
+ runId,
2251
+ budget_usd: effectiveBudget,
2252
+ cost_captured: false
2253
+ });
2254
+ await waitForHook(hooks.onAfterAnalyze ? () => hooks.onAfterAnalyze?.({
2255
+ analyst,
2256
+ summary,
2257
+ findings: hookFindings,
2258
+ runId
2259
+ }) : void 0, runSignal);
2260
+ yield {
2261
+ type: "analyst-completed",
2262
+ summary,
2263
+ findings: hookFindings
2264
+ };
2265
+ continue;
2266
+ }
2267
+ await waitForHook(hooks.onAfterAnalyze ? () => hooks.onAfterAnalyze?.({
2268
+ analyst,
2269
+ summary,
2270
+ findings,
2271
+ runId
2272
+ }) : void 0, runSignal);
2273
+ yield {
2274
+ type: "analyst-completed",
2275
+ summary,
2276
+ findings
2277
+ };
2278
+ }
2279
+ const result = {
2280
+ run_id: runId,
2281
+ correlation_id: correlationId,
2282
+ started_at: startedAt,
2283
+ ended_at: (/* @__PURE__ */ new Date()).toISOString(),
2284
+ findings: allFindings,
2285
+ per_analyst: summaries,
2286
+ total_cost_usd: totalCost,
2287
+ total_cost_provenance: aggregateCostProvenance(summaries.map((summary) => summary.usage?.cost ?? {
2288
+ kind: "uncaptured",
2289
+ usd: null
2290
+ }))
2291
+ };
2292
+ await waitForHook(hooks.onComplete ? () => hooks.onComplete?.({ result }) : void 0, runSignal);
2293
+ yield {
2294
+ type: "run-completed",
2295
+ result
2296
+ };
2297
+ }
2298
+ selectAnalysts(opts) {
2299
+ let candidates = Array.from(this.analysts.values());
2300
+ if (opts.only?.length) {
2301
+ const only = new Set(opts.only);
2302
+ candidates = candidates.filter((a) => only.has(a.id));
2303
+ }
2304
+ if (opts.skip?.length) {
2305
+ const skip = new Set(opts.skip);
2306
+ candidates = candidates.filter((a) => !skip.has(a.id));
2307
+ }
2308
+ return candidates;
2309
+ }
2310
+ routeInput(analyst, inputs) {
2311
+ switch (analyst.inputKind) {
2312
+ case "trace-store": return inputs.traceStore ? {
2313
+ kind: "present",
2314
+ value: inputs.traceStore
2315
+ } : { kind: "missing" };
2316
+ case "artifact-dir": return inputs.artifactDir ? {
2317
+ kind: "present",
2318
+ value: inputs.artifactDir
2319
+ } : { kind: "missing" };
2320
+ case "run-record": return inputs.runRecord ? {
2321
+ kind: "present",
2322
+ value: inputs.runRecord
2323
+ } : { kind: "missing" };
2324
+ case "judge-input": return inputs.judgeInput ? {
2325
+ kind: "present",
2326
+ value: inputs.judgeInput
2327
+ } : { kind: "missing" };
2328
+ case "custom": {
2329
+ const v = inputs.custom?.[analyst.id];
2330
+ return v !== void 0 ? {
2331
+ kind: "present",
2332
+ value: v
2333
+ } : { kind: "missing" };
2334
+ }
2335
+ }
2336
+ }
2337
+ };
2338
+ function validateTimeout(timeoutMs) {
2339
+ if (timeoutMs === void 0) return void 0;
2340
+ if (!Number.isSafeInteger(timeoutMs) || timeoutMs <= 0 || timeoutMs > 2147483647) throw new TypeError("RegistryRunOpts.timeoutMs must be a positive safe integer no greater than 2147483647");
2341
+ return timeoutMs;
2342
+ }
2343
+ async function waitForOperation(operation, signal, abortGraceMs) {
2344
+ if (!signal) return operation;
2345
+ if (signal.aborted) {
2346
+ operation.catch(() => {});
2347
+ throw abortReason(signal);
2348
+ }
2349
+ return new Promise((resolve, reject) => {
2350
+ let settlementTimer;
2351
+ const cleanup = () => {
2352
+ signal.removeEventListener("abort", onAbort);
2353
+ if (settlementTimer) clearTimeout(settlementTimer);
2354
+ };
2355
+ const onAbort = () => {
2356
+ if (abortGraceMs === 0) {
2357
+ cleanup();
2358
+ reject(abortReason(signal));
2359
+ return;
2360
+ }
2361
+ settlementTimer = setTimeout(() => {
2362
+ cleanup();
2363
+ reject(abortReason(signal));
2364
+ }, abortGraceMs);
2365
+ };
2366
+ signal.addEventListener("abort", onAbort, { once: true });
2367
+ operation.then((value) => {
2368
+ cleanup();
2369
+ if (signal.aborted) reject(abortReason(signal));
2370
+ else resolve(value);
2371
+ }, (error) => {
2372
+ cleanup();
2373
+ reject(signal.aborted ? abortReason(signal) : error);
2374
+ });
2375
+ });
2376
+ }
2377
+ async function waitForHook(operation, signal) {
2378
+ if (operation === void 0 || signal?.aborted) return void 0;
2379
+ try {
2380
+ return await waitForOperation(Promise.resolve().then(() => {
2381
+ if (signal?.aborted) throw abortReason(signal);
2382
+ return operation();
2383
+ }), signal, 0);
2384
+ } catch (error) {
2385
+ if (signal?.aborted) return void 0;
2386
+ throw error;
2387
+ }
2388
+ }
2389
+ function analystAbortGraceMs(analyst) {
2390
+ if (analyst.cost.kind === "deterministic") return 0;
2391
+ const settlementMs = validateUsageSettlementTimeout(analyst.cost.settlement_timeout_ms);
2392
+ if (settlementMs === 0) return 0;
2393
+ return Math.min(settlementMs + 100, 2147483647);
2394
+ }
2395
+ function abortedBeforeStartSummary(analyst, signal, latencyMs = 0) {
2396
+ const reason = abortReason(signal);
2397
+ return {
2398
+ analyst_id: analyst.id,
2399
+ status: "skipped",
2400
+ reason: `${reason.name}: ${reason.message}`,
2401
+ findings_count: 0,
2402
+ latency_ms: latencyMs,
2403
+ usage: zeroUsage()
2404
+ };
2405
+ }
2406
+ function abortReason(signal) {
2407
+ return signal.reason instanceof Error ? signal.reason : new DOMException("The operation was aborted", "AbortError");
2408
+ }
2409
+ /**
2410
+ * Default budget allocator: prefer the custom `allocate` callback if
2411
+ * provided; else weighted split when weights are set; else equal split
2412
+ * across `runningCount`. Returns undefined when no totalUsd is known.
2413
+ */
2414
+ function allocateBudget(policy, args) {
2415
+ if (!policy) return void 0;
2416
+ if (policy.allocate) {
2417
+ const allocated = policy.allocate({
2418
+ analyst: args.analyst,
2419
+ totalUsd: policy.totalUsd,
2420
+ remainingUsd: args.remainingUsd,
2421
+ runningCount: args.runningCount
2422
+ });
2423
+ if (allocated === void 0) {
2424
+ if (policy.totalUsd !== void 0) throw new Error(`BudgetPolicy.allocate('${args.analyst.id}') cannot return undefined when totalUsd is set`);
2425
+ return;
2426
+ }
2427
+ assertBudgetAmount(allocated, `BudgetPolicy.allocate('${args.analyst.id}')`);
2428
+ return args.remainingUsd === void 0 ? allocated : Math.min(allocated, args.remainingUsd);
2429
+ }
2430
+ if (policy.totalUsd == null) return void 0;
2431
+ const allocated = policy.weights ? policy.totalUsd * analystWeight(policy.weights, args.analyst.id) / args.totalWeight : policy.totalUsd / Math.max(1, args.runningCount);
2432
+ return args.remainingUsd === void 0 ? allocated : Math.min(allocated, args.remainingUsd);
2433
+ }
2434
+ function validateBudgetPolicy(policy) {
2435
+ if (!policy) return;
2436
+ if (policy.totalUsd !== void 0) assertBudgetAmount(policy.totalUsd, "BudgetPolicy.totalUsd");
2437
+ for (const [analystId, weight] of Object.entries(policy.weights ?? {})) assertBudgetAmount(weight, `BudgetPolicy.weights['${analystId}']`);
2438
+ }
2439
+ function assertBudgetAmount(value, field) {
2440
+ if (!Number.isFinite(value) || value < 0) throw new Error(`${field} must be a non-negative finite number`);
2441
+ }
2442
+ function validateEffectiveBudget(budgetUsd, remainingUsd, analystId) {
2443
+ if (budgetUsd !== void 0) assertBudgetAmount(budgetUsd, `AnalystContext.budgetUsd for '${analystId}'`);
2444
+ if (remainingUsd === void 0) return budgetUsd;
2445
+ if (budgetUsd === void 0) throw new Error(`AnalystContext.budgetUsd for '${analystId}' cannot be removed while an overall budget remains`);
2446
+ if (budgetUsd > remainingUsd) throw new Error(`AnalystContext.budgetUsd for '${analystId}' (${budgetUsd}) exceeds the remaining overall budget (${remainingUsd})`);
2447
+ return budgetUsd;
2448
+ }
2449
+ function analystWeight(weights, analystId) {
2450
+ const weight = weights[analystId] ?? 1;
2451
+ assertBudgetAmount(weight, `BudgetPolicy.weights['${analystId}']`);
2452
+ return weight;
2453
+ }
2454
+ function zeroUsage() {
2455
+ return {
2456
+ calls: 0,
2457
+ tokens: {
2458
+ input: 0,
2459
+ output: 0
2460
+ },
2461
+ cost: {
2462
+ kind: "observed",
2463
+ usd: 0
2464
+ }
2465
+ };
2466
+ }
2467
+ function resolveUsage(analyst, receipts) {
2468
+ if (receipts.length > 0) return mergeUsageReceipts(receipts);
2469
+ if (analyst.cost.kind === "deterministic") return zeroUsage();
2470
+ return {
2471
+ calls: null,
2472
+ tokens: null,
2473
+ cost: {
2474
+ kind: "uncaptured",
2475
+ usd: null
2476
+ }
2477
+ };
2478
+ }
2479
+ function mergeUsageReceipts(receipts) {
2480
+ const calls = receipts.every((receipt) => receipt.calls !== null) ? receipts.reduce((sum, receipt) => sum + (receipt.calls ?? 0), 0) : null;
2481
+ const tokens = receipts.every((receipt) => receipt.tokens !== null) ? receipts.reduce((sum, receipt) => ({
2482
+ input: sum.input + (receipt.tokens?.input ?? 0),
2483
+ output: sum.output + (receipt.tokens?.output ?? 0),
2484
+ ...sum.reasoning !== void 0 || receipt.tokens?.reasoning !== void 0 ? { reasoning: (sum.reasoning ?? 0) + (receipt.tokens?.reasoning ?? 0) } : {},
2485
+ ...sum.cached !== void 0 || receipt.tokens?.cached !== void 0 ? { cached: (sum.cached ?? 0) + (receipt.tokens?.cached ?? 0) } : {},
2486
+ ...sum.cacheWrite !== void 0 || receipt.tokens?.cacheWrite !== void 0 ? { cacheWrite: (sum.cacheWrite ?? 0) + (receipt.tokens?.cacheWrite ?? 0) } : {}
2487
+ }), {
2488
+ input: 0,
2489
+ output: 0
2490
+ }) : null;
2491
+ const cost = aggregateCostProvenance(receipts.map((receipt) => receipt.cost));
2492
+ return {
2493
+ calls,
2494
+ tokens,
2495
+ cost,
2496
+ ...cost.kind === "uncaptured" ? { knownCostUsd: receipts.reduce((sum, receipt) => sum + knownCostUsd(receipt), 0) } : {}
2497
+ };
2498
+ }
2499
+ function knownCostUsd(receipt) {
2500
+ return receipt.cost.kind === "uncaptured" ? receipt.knownCostUsd ?? 0 : receipt.cost.usd;
2501
+ }
2502
+ function budgetDebit(receipt, allocatedUsd) {
2503
+ const known = knownCostUsd(receipt);
2504
+ return receipt.cost.kind === "uncaptured" && allocatedUsd !== void 0 ? Math.max(known, allocatedUsd) : known;
2505
+ }
2506
+ function aggregateCostProvenance(costs) {
2507
+ if (costs.some((cost) => cost.kind === "uncaptured")) return {
2508
+ kind: "uncaptured",
2509
+ usd: null
2510
+ };
2511
+ const usd = costs.reduce((sum, cost) => sum + (cost.usd ?? 0), 0);
2512
+ return costs.some((cost) => cost.kind === "estimated") ? {
2513
+ kind: "estimated",
2514
+ usd
2515
+ } : {
2516
+ kind: "observed",
2517
+ usd
2518
+ };
2519
+ }
2520
+ function assertValidUsageReceipt(receipt) {
2521
+ if (receipt.calls !== null && (!Number.isInteger(receipt.calls) || receipt.calls < 0)) throw new Error("AnalystContext.recordUsage: calls must be a non-negative integer or null");
2522
+ if (receipt.tokens) {
2523
+ assertNonNegativeFinite(receipt.tokens.input, "tokens.input");
2524
+ assertNonNegativeFinite(receipt.tokens.output, "tokens.output");
2525
+ if (receipt.tokens.reasoning !== void 0) {
2526
+ assertNonNegativeFinite(receipt.tokens.reasoning, "tokens.reasoning");
2527
+ if (receipt.tokens.reasoning > receipt.tokens.output) throw new Error("AnalystContext.recordUsage: tokens.reasoning must not exceed tokens.output");
2528
+ }
2529
+ if (receipt.tokens.cached !== void 0) assertNonNegativeFinite(receipt.tokens.cached, "tokens.cached");
2530
+ if (receipt.tokens.cacheWrite !== void 0) assertNonNegativeFinite(receipt.tokens.cacheWrite, "tokens.cacheWrite");
2531
+ }
2532
+ if (receipt.cost.kind !== "uncaptured") assertNonNegativeFinite(receipt.cost.usd, "cost.usd");
2533
+ else if (receipt.cost.usd !== null) throw new Error("AnalystContext.recordUsage: uncaptured cost.usd must be null");
2534
+ if (receipt.knownCostUsd !== void 0) assertNonNegativeFinite(receipt.knownCostUsd, "knownCostUsd");
2535
+ }
2536
+ function assertNonNegativeFinite(value, field) {
2537
+ if (!Number.isFinite(value) || value < 0) throw new Error(`AnalystContext.recordUsage: ${field} must be a non-negative finite number`);
2538
+ }
2539
+ /**
2540
+ * Resolve the `priorFindings` slice an analyst sees.
2541
+ *
2542
+ * - Array form → the analyst sees only findings whose `analyst_id`
2543
+ * matches its own id, so a kind never reads
2544
+ * another kind's history by accident.
2545
+ * - Record form → the analyst gets the entry keyed by its id, with
2546
+ * the `'*'` wildcard appended (in that order). Use
2547
+ * the wildcard when several kinds should see the same
2548
+ * historical findings.
2549
+ */
2550
+ function selectPriorFindings(source, analystId) {
2551
+ if (!source) return void 0;
2552
+ if (Array.isArray(source)) {
2553
+ const own = source.filter((f) => f.analyst_id === analystId);
2554
+ return own.length > 0 ? own : void 0;
2555
+ }
2556
+ const record = source;
2557
+ const own = record[analystId] ?? [];
2558
+ const wildcard = record["*"] ?? [];
2559
+ const merged = [...own, ...wildcard];
2560
+ return merged.length > 0 ? merged : void 0;
2561
+ }
2562
+ //#endregion
2563
+ //#region src/analyst/default-registry.ts
2564
+ function buildDefaultAnalystRegistry(opts = {}) {
2565
+ const registry = new AnalystRegistry(opts.registry);
2566
+ if (opts.includeBehavioral !== false) registry.register(behavioralAnalyst());
2567
+ if (opts.ai) {
2568
+ const kinds = opts.kinds ?? DEFAULT_TRACE_ANALYST_KINDS;
2569
+ for (const spec of kinds) registry.register(createTraceAnalystKind(spec, {
2570
+ ai: opts.ai,
2571
+ model: opts.model
2572
+ }));
2573
+ }
2574
+ return registry;
2575
+ }
2576
+ //#endregion
2577
+ export { parseFindingSubject as A, stripCodeFences as C, FindingSubjectStringSchema as D, FINDING_SUBJECT_SYNTAX as E, makeFinding as F, computeTraceMetrics as I, createChatClient as L, behavioralAnalyst as M, deriveEfficiencyFindings as N, KIND_EXPECTED_SUBJECTS as O, computeFindingId as P, createAnalystAi as R, coerceToFindingRows as S, FINDING_SUBJECT_KINDS as T, RawAnalystEvidenceSchema as _, KNOWLEDGE_GAP_KIND_SPEC as a, parseRawFinding as b, buildTraceToolsForGroup as c, renderUpstreamFindings as d, settleUsageReceiptFromCostLedger as f, RAW_FINDING_SCHEMA_PROMPT as g, ANALYST_SEVERITIES as h, KNOWLEDGE_POISONING_KIND_SPEC as i, renderFindingSubject as j, findingSubjectGrammarPromptFor as k, createTraceAnalystKind as l, structureFindings as m, AnalystRegistry as n, IMPROVEMENT_KIND_SPEC as o, validateUsageSettlementTimeout as p, DEFAULT_TRACE_ANALYST_KINDS as r, FAILURE_MODE_KIND_SPEC as s, buildDefaultAnalystRegistry as t, renderPriorFindings as u, RawAnalystFindingSchema as v, FINDING_SUBJECT_GRAMMAR_PROMPT as w, coerceJson as x, evidenceRefsFromRawFinding as y };
2578
+
2579
+ //# sourceMappingURL=default-registry-C-vFCSEc.js.map