@tangle-network/agent-eval 0.129.0 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (427) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +1 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/package.json +17 -9
  301. package/dist/benchmarks/index.js.map +0 -1
  302. package/dist/campaign/index.js.map +0 -1
  303. package/dist/chunk-2QU3YOPR.js +0 -7374
  304. package/dist/chunk-2QU3YOPR.js.map +0 -1
  305. package/dist/chunk-3OCR4R5I.js +0 -728
  306. package/dist/chunk-3OCR4R5I.js.map +0 -1
  307. package/dist/chunk-3RF76KTD.js +0 -84
  308. package/dist/chunk-3RF76KTD.js.map +0 -1
  309. package/dist/chunk-56TAVBOK.js +0 -698
  310. package/dist/chunk-5DTSBUL2.js +0 -159
  311. package/dist/chunk-5DTSBUL2.js.map +0 -1
  312. package/dist/chunk-7FO3TNPI.js +0 -232
  313. package/dist/chunk-7FO3TNPI.js.map +0 -1
  314. package/dist/chunk-7ZZMD7UK.js +0 -386
  315. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  316. package/dist/chunk-BOD4O7OF.js +0 -40
  317. package/dist/chunk-BOD4O7OF.js.map +0 -1
  318. package/dist/chunk-BSO5JDQH.js +0 -2335
  319. package/dist/chunk-BSO5JDQH.js.map +0 -1
  320. package/dist/chunk-C6LXANRU.js +0 -1550
  321. package/dist/chunk-C6LXANRU.js.map +0 -1
  322. package/dist/chunk-DODXQREJ.js +0 -752
  323. package/dist/chunk-DODXQREJ.js.map +0 -1
  324. package/dist/chunk-DRYIUNWY.js +0 -622
  325. package/dist/chunk-DRYIUNWY.js.map +0 -1
  326. package/dist/chunk-E7QXT7SX.js +0 -183
  327. package/dist/chunk-E7QXT7SX.js.map +0 -1
  328. package/dist/chunk-EG66UGL4.js +0 -341
  329. package/dist/chunk-EG66UGL4.js.map +0 -1
  330. package/dist/chunk-FXTVJPYD.js +0 -576
  331. package/dist/chunk-FXTVJPYD.js.map +0 -1
  332. package/dist/chunk-G7MGMCZD.js +0 -153
  333. package/dist/chunk-G7MGMCZD.js.map +0 -1
  334. package/dist/chunk-GGE4NNQT.js +0 -65
  335. package/dist/chunk-GGE4NNQT.js.map +0 -1
  336. package/dist/chunk-H23X7XKK.js +0 -181
  337. package/dist/chunk-H23X7XKK.js.map +0 -1
  338. package/dist/chunk-HHWE3POT.js +0 -94
  339. package/dist/chunk-HHWE3POT.js.map +0 -1
  340. package/dist/chunk-HPWUNB47.js +0 -289
  341. package/dist/chunk-HPWUNB47.js.map +0 -1
  342. package/dist/chunk-IYCLP2N2.js +0 -766
  343. package/dist/chunk-IYCLP2N2.js.map +0 -1
  344. package/dist/chunk-JHCHEVET.js +0 -274
  345. package/dist/chunk-JHCHEVET.js.map +0 -1
  346. package/dist/chunk-JQSF5DQT.js +0 -701
  347. package/dist/chunk-JQSF5DQT.js.map +0 -1
  348. package/dist/chunk-K4DBDHLK.js +0 -158
  349. package/dist/chunk-K4DBDHLK.js.map +0 -1
  350. package/dist/chunk-K6N6XJJX.js +0 -306
  351. package/dist/chunk-K6N6XJJX.js.map +0 -1
  352. package/dist/chunk-M4YBQKIJ.js +0 -1040
  353. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  354. package/dist/chunk-MA6HLL3S.js +0 -65
  355. package/dist/chunk-MA6HLL3S.js.map +0 -1
  356. package/dist/chunk-MAZ26DC7.js +0 -99
  357. package/dist/chunk-MAZ26DC7.js.map +0 -1
  358. package/dist/chunk-NPCTHQIO.js +0 -91
  359. package/dist/chunk-NPCTHQIO.js.map +0 -1
  360. package/dist/chunk-NY44NC4A.js +0 -1056
  361. package/dist/chunk-NY44NC4A.js.map +0 -1
  362. package/dist/chunk-OIUOT4QD.js +0 -44
  363. package/dist/chunk-OIUOT4QD.js.map +0 -1
  364. package/dist/chunk-ONWEPEDO.js +0 -57
  365. package/dist/chunk-ONWEPEDO.js.map +0 -1
  366. package/dist/chunk-OWN5NPMC.js +0 -152
  367. package/dist/chunk-OWN5NPMC.js.map +0 -1
  368. package/dist/chunk-P6FYH6K4.js +0 -1161
  369. package/dist/chunk-P6FYH6K4.js.map +0 -1
  370. package/dist/chunk-PC4UYEBM.js +0 -166
  371. package/dist/chunk-PC4UYEBM.js.map +0 -1
  372. package/dist/chunk-PC5DOSM7.js +0 -579
  373. package/dist/chunk-PC5DOSM7.js.map +0 -1
  374. package/dist/chunk-PXE2VKMX.js +0 -140
  375. package/dist/chunk-PXE2VKMX.js.map +0 -1
  376. package/dist/chunk-PZ5AY32C.js +0 -10
  377. package/dist/chunk-PZ5AY32C.js.map +0 -1
  378. package/dist/chunk-QB6BDBP2.js +0 -4464
  379. package/dist/chunk-QB6BDBP2.js.map +0 -1
  380. package/dist/chunk-RXHCETDZ.js +0 -536
  381. package/dist/chunk-RXHCETDZ.js.map +0 -1
  382. package/dist/chunk-RZTMDUO7.js +0 -49
  383. package/dist/chunk-RZTMDUO7.js.map +0 -1
  384. package/dist/chunk-SFLLL76A.js +0 -669
  385. package/dist/chunk-SFLLL76A.js.map +0 -1
  386. package/dist/chunk-SZLVEKMJ.js +0 -1446
  387. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  388. package/dist/chunk-T4SQEITX.js +0 -95
  389. package/dist/chunk-T4SQEITX.js.map +0 -1
  390. package/dist/chunk-T6RLYGAD.js +0 -158
  391. package/dist/chunk-T6RLYGAD.js.map +0 -1
  392. package/dist/chunk-TJVT4QFF.js +0 -911
  393. package/dist/chunk-TJVT4QFF.js.map +0 -1
  394. package/dist/chunk-TQ7LNKZ3.js +0 -136
  395. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  396. package/dist/chunk-U4L7JRPZ.js +0 -1706
  397. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  398. package/dist/chunk-U4PHLT2N.js +0 -419
  399. package/dist/chunk-U4PHLT2N.js.map +0 -1
  400. package/dist/chunk-VCZ5FQYW.js +0 -928
  401. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  402. package/dist/chunk-VI2UW6B6.js +0 -162
  403. package/dist/chunk-VI2UW6B6.js.map +0 -1
  404. package/dist/chunk-VQMK5FMP.js +0 -247
  405. package/dist/chunk-VQMK5FMP.js.map +0 -1
  406. package/dist/chunk-WGXIEX7P.js +0 -116
  407. package/dist/chunk-WGXIEX7P.js.map +0 -1
  408. package/dist/chunk-WVATSFCP.js +0 -1553
  409. package/dist/chunk-WVATSFCP.js.map +0 -1
  410. package/dist/chunk-X4YIBDER.js +0 -1662
  411. package/dist/chunk-X4YIBDER.js.map +0 -1
  412. package/dist/chunk-YQN4ICPP.js +0 -355
  413. package/dist/chunk-YQN4ICPP.js.map +0 -1
  414. package/dist/chunk-ZET2UAYW.js +0 -89
  415. package/dist/chunk-ZET2UAYW.js.map +0 -1
  416. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  417. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  418. package/dist/control.js.map +0 -1
  419. package/dist/hosted/index.js.map +0 -1
  420. package/dist/matrix/index.js.map +0 -1
  421. package/dist/reporting.js.map +0 -1
  422. package/dist/rollout/index.js.map +0 -1
  423. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  424. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  425. package/dist/supervisor-run/index.js.map +0 -1
  426. package/dist/traces.js.map +0 -1
  427. package/dist/wire/index.js.map +0 -1
@@ -1,2834 +1,67 @@
1
- import { AxAIArgs, AxAIService, AxFunction } from '@ax-llm/ax';
2
- import { z } from 'zod';
3
-
4
- /**
5
- * Validator-output verdict substrate primitive for "did this output pass,
6
- * and how well?"
7
- *
8
- * Used by:
9
- * - `@tangle-network/agent-eval/matrix` — verdict per cell in the cartesian.
10
- * - `@tangle-network/agent-runtime` — Validator<Output, Verdict = DefaultVerdict>.
11
- * Runtime keeps `Validator` because it's coupled to runtime-shaped
12
- * `ValidationCtx` (iteration, signal, traceEmitter); the verdict TYPE
13
- * itself is a substrate concept and lives here.
14
- *
15
- * Repo layering: agent-eval is the substrate (no upward deps). Both
16
- * agent-runtime and agent-knowledge consume this type FROM agent-eval —
17
- * never the other way around. See CLAUDE.md "Repo layering" for the rule.
18
- */
19
- /**
20
- * Minimal verdict shape — `valid` + `score` are required; `scores` +
21
- * `notes` are optional surface. Validators that need richer shapes
22
- * parameterise `Validator<Output, MyVerdict>` with their own type.
23
- *
24
- * Need structured extras? Extend DefaultVerdict with typed fields — never
25
- * serialize extras into `notes`.
26
- */
27
- interface DefaultVerdict {
28
- /** Whether the output meets the validator's pass criteria. */
29
- valid: boolean;
30
- /** Aggregate score in [0, 1]. Drivers use this for winner selection. */
31
- score: number;
32
- /** Per-dimension scores. Free-form; weighted into `score` by the validator. */
33
- scores?: Record<string, number>;
34
- /** Human-readable rationale; surfaces in trace + final-result `winner.verdict`. */
35
- notes?: string;
36
- }
37
-
38
- /**
39
- * Multi-layer verifier — ordered pipeline of verification layers.
40
- *
41
- * Different contract from {@link JudgeRunner} (which runs parallel
42
- * specs against a sandbox). MultiLayerVerifier is a DAG of layers
43
- * (install → typecheck → build → lint → serve → semantic → …) with
44
- * dependency-based skip, per-layer findings, soft-fail semantics, and
45
- * an aggregated `blendedScore` across all passed layers.
46
- *
47
- * Use when you want:
48
- * - ordered stages where a failing upstream stage skips downstream ones
49
- * - each stage produces rich `findings` (severity + message + evidence)
50
- * - a single composite score across stages with per-stage weights
51
- * - soft-fail stages whose failure doesn't abort the pipeline
52
- *
53
- * Use {@link JudgeRunner} when you want:
54
- * - N independent judges running in parallel against the same artifact
55
- * - no inter-judge dependencies
56
- * - boolean `passed` per judge + overall
57
- *
58
- * Both primitives compose — JudgeRunner can be invoked as a single
59
- * layer inside a MultiLayerVerifier if that suits the caller.
60
- */
61
-
62
- type LayerStatus = 'pass' | 'fail' | 'skipped' | 'error' | 'timeout';
63
- type Severity = 'critical' | 'major' | 'minor' | 'info';
64
- interface Finding {
65
- severity: Severity;
66
- message: string;
67
- evidence?: string;
68
- /** Optional layer name the finding belongs to (set by the verifier if omitted). */
69
- layer?: string;
70
- /**
71
- * Free-form structured payload — used by `multiToolchainLayer` to attach
72
- * `{ adapter: 'pnpm' }`, by judges to attach evidence pointers, etc.
73
- * Renderers MAY interrogate; agent-eval primitives never assume shape.
74
- */
75
- detail?: Record<string, unknown>;
76
- }
77
- interface LayerResult {
78
- layer: string;
79
- status: LayerStatus;
80
- /** Origin of an `error` or `timeout`. Defaults to `execution`. */
81
- errorSource?: 'execution' | 'judge';
82
- /** 0..1 score, optional — layers that don't produce a numeric score omit. */
83
- score?: number;
84
- durationMs: number;
85
- findings: Finding[];
86
- /** Short human-readable summary (one line). */
87
- reason?: string;
88
- /**
89
- * Numeric layer-level diagnostics: error counts, warning counts,
90
- * cyclomatic complexity, total adapter wall-time, etc. Keyed by
91
- * diagnostic name; null = "diagnostic not applicable / not measured."
92
- * Renderers that know the keys can display them; ones that don't,
93
- * ignore. Free-form on purpose — consumers type the value shape in
94
- * their own namespace.
95
- */
96
- diagnostics?: Record<string, number | null>;
97
- /** Any rich per-layer detail — rendered as-is by consumers that know the layer. */
98
- detail?: Record<string, unknown>;
99
- }
100
- interface VerifyContext<Env = unknown> {
101
- /** Per-run opaque context the caller provides. Layers destructure what they need. */
102
- env: Env;
103
- /** Previously-computed results from layers that already ran. */
104
- prior: Record<string, LayerResult>;
105
- /** Signal — if aborted, layers MUST bail within reasonable wall. */
106
- signal: AbortSignal;
107
- }
108
- interface Layer<Env = unknown> {
109
- name: string;
110
- /** Origin assigned when this layer errors or times out. Defaults to `execution`. */
111
- errorSource?: 'execution' | 'judge';
112
- /** Stages that must have `status: 'pass'` before this layer runs. */
113
- dependsOn?: string[];
114
- /**
115
- * Weight in the composite `blendedScore`. Default 1.0. Layers with weight 0
116
- * contribute findings but not score.
117
- */
118
- weight?: number;
119
- /**
120
- * If true, a `fail` status contributes to `blendedScore` (as 0) instead of
121
- * being dropped — use for layers whose failure is a real signal. Default:
122
- * fail drops from numerator + denominator, matching VB's existing semantics.
123
- */
124
- failContributesToScore?: boolean;
125
- /** Optional per-layer wall-cap in ms. Honored by the verifier (AbortSignal). */
126
- capMs?: number;
127
- run: (ctx: VerifyContext<Env>) => Promise<LayerResult> | LayerResult;
128
- }
129
- interface VerifyOptions<Env = unknown> {
130
- env: Env;
131
- /**
132
- * Overall wall cap. Default: sum of layer capMs, or Infinity if any layer
133
- * omits a cap. The verifier short-circuits remaining layers on overall cap.
134
- */
135
- overallCapMs?: number;
136
- /** Called with each layer result as it completes. */
137
- onLayer?: (result: LayerResult) => void;
138
- }
139
- /** Extends the substrate verdict spine: `valid` = `allPass`; `score` is the
140
- * complete task score or 0 when the configured scoring panel was incomplete. */
141
- interface VerificationReport extends DefaultVerdict {
142
- layers: LayerResult[];
143
- passCount: number;
144
- failCount: number;
145
- skippedCount: number;
146
- errorCount: number;
147
- /** True iff the configured scoring panel completed and every layer passed. */
148
- allPass: boolean;
149
- /**
150
- * Diagnostic weighted mean across contributing layers. This may represent a
151
- * partial panel. It is 0 when no layer contributed.
152
- */
153
- blendedScore: number;
154
- /**
155
- * Complete task-quality measurement.
156
- * Present when at least one layer produced a valid score, every other layer
157
- * completed successfully or contributed an explicit scored failure, and no
158
- * result is missing because of a failure, skip, error, or timeout.
159
- * Use this field, not `blendedScore`, when creating task labels.
160
- */
161
- taskScore?: number;
162
- durationMs: number;
163
- startedAt: string;
164
- finishedAt: string;
165
- }
166
- /**
167
- * Ordered DAG of verification layers with dependency-based skipping, per-layer findings, soft-fail semantics, and a blended composite score across all passed layers.
168
- */
169
- declare class MultiLayerVerifier<Env = unknown> {
170
- private readonly layers;
171
- constructor(layers: Layer<Env>[]);
172
- run(opts: VerifyOptions<Env>): Promise<VerificationReport>;
173
- }
174
-
175
- interface RunScore {
176
- success: number;
177
- goalProgress: number;
178
- repoGroundedness: number;
179
- driftPenalty: number;
180
- toolUseQuality: number;
181
- patchQuality: number;
182
- testReality: number;
183
- finalGate: number;
184
- reviewerBlockers: number;
185
- costUsd: number;
186
- wallSeconds: number;
187
- notes?: string[];
188
- }
189
- interface RunScoreWeights {
190
- success: number;
191
- goalProgress: number;
192
- repoGroundedness: number;
193
- driftPenalty: number;
194
- toolUseQuality: number;
195
- patchQuality: number;
196
- testReality: number;
197
- finalGate: number;
198
- reviewerBlockers: number;
199
- costUsd: number;
200
- wallSeconds: number;
201
- }
202
-
203
- type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
204
- type AgentProfileDimensionValue = string | number | boolean | null;
205
- interface AgentProfileSource {
206
- /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
207
- kind: string;
208
- /** sha256 over the canonical source profile object. */
209
- hash: string;
210
- }
211
- interface AgentProfileHarness {
212
- id: string;
213
- version?: string;
214
- hash?: string;
215
- }
216
- interface AgentProfileCell {
217
- schemaVersion: AgentProfileCellSchemaVersion;
218
- cellId: string;
219
- profileId: string;
220
- sourceProfile: AgentProfileSource;
221
- harness?: AgentProfileHarness;
222
- model?: string;
223
- promptHash?: string;
224
- dimensions?: Record<string, AgentProfileDimensionValue>;
225
- }
226
-
227
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
228
- interface BudgetSpec {
229
- tokens?: number;
230
- wallMs?: number;
231
- calls?: number;
232
- usd?: number;
233
- }
234
- interface RunOutcome$1 {
235
- score?: number;
236
- pass?: boolean;
237
- failureClass?: FailureClass;
238
- notes?: string;
239
- }
240
- /**
241
- * Layer — optional classification in a nested build workflow.
242
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
243
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
244
- * `app-runtime`: a run of the generated agent against a domain scenario.
245
- * `meta`: any meta-eval (judge replay, correlation analysis).
246
- */
247
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
248
- interface Run {
249
- runId: string;
250
- /**
251
- * Stable identifier of the scenario being executed.
252
- *
253
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
254
- * input WITHOUT this field, substituting a sensible default
255
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
256
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
257
- * keeps the persisted shape unambiguous for downstream filters + aggregations
258
- * while removing the boilerplate of inventing placeholder ids at the call site.
259
- */
260
- scenarioId: string;
261
- variantId?: string;
262
- datasetVersion?: string;
263
- /** Git SHA of agent code at run time. */
264
- codeSha?: string;
265
- /** Hash of the prompt template + any system prompt. */
266
- promptSha?: string;
267
- /** Model id + date + system-prompt hash, concatenated. */
268
- modelFingerprint?: string;
269
- seed?: number;
270
- /** Arbitrary environment markers (shell, docker version, tz). */
271
- envFingerprint?: Record<string, string>;
272
- /** Version of the redaction rules applied to this run. */
273
- redactionVersion?: string;
274
- /** Parent run in a nested build workflow. A builder run's children are
275
- * app-build runs; those children are app-runtime runs. */
276
- parentRunId?: string;
277
- /** Stable project identifier — groups runs across chats + sessions. */
278
- projectId?: string;
279
- /** Chat/conversation identifier within a project. */
280
- chatId?: string;
281
- /** Layer classification — hint for aggregation; not enforced. */
282
- layer?: RunLayer;
283
- startedAt: number;
284
- endedAt?: number;
285
- status: RunStatus;
286
- outcome?: RunOutcome$1;
287
- budget?: BudgetSpec;
288
- /** Free-form labels for downstream grouping. */
289
- tags?: Record<string, string>;
290
- }
291
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
292
- type SpanStatus = 'ok' | 'error';
293
- interface SpanBase {
294
- spanId: string;
295
- parentSpanId?: string;
296
- runId: string;
297
- kind: SpanKind;
298
- name: string;
299
- startedAt: number;
300
- endedAt?: number;
301
- status?: SpanStatus;
302
- error?: string;
303
- /** Anything not covered by typed fields. Kept deliberately free-form. */
304
- attributes?: Record<string, unknown>;
305
- }
306
- interface Message {
307
- role: 'system' | 'user' | 'assistant' | 'tool';
308
- content: string;
309
- tokens?: number;
310
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
311
- images?: Array<{
312
- artifactId?: string;
313
- url?: string;
314
- mime?: string;
315
- }>;
316
- }
317
- interface LlmSpan extends SpanBase {
318
- kind: 'llm';
319
- model: string;
320
- messages: Message[];
321
- output?: string;
322
- inputTokens?: number;
323
- /** All generated tokens, including the reasoning subset when present. */
324
- outputTokens?: number;
325
- cachedTokens?: number;
326
- cacheWriteTokens?: number;
327
- /** Reasoning-token subset of `outputTokens`. */
328
- reasoningTokens?: number;
329
- costUsd?: number;
330
- finishReason?: string;
331
- }
332
- interface ToolSpan extends SpanBase {
333
- kind: 'tool';
334
- toolName: string;
335
- args: unknown;
336
- /** False when the source observed the call but did not capture its arguments. */
337
- argsCaptured?: boolean;
338
- result?: unknown;
339
- latencyMs?: number;
340
- }
341
- interface RetrievalSpan extends SpanBase {
342
- kind: 'retrieval';
343
- query: string;
344
- hits: Array<{
345
- docId: string;
346
- score: number;
347
- content?: string;
348
- }>;
349
- }
350
- interface JudgeSpan extends SpanBase {
351
- kind: 'judge';
352
- judgeId: string;
353
- /** Span this judgment applies to. */
354
- targetSpanId: string;
355
- dimension: string;
356
- /** Numeric score (free-range; interpretation up to the judge). */
357
- score: number;
358
- rationale?: string;
359
- evidence?: string;
360
- }
361
- interface SandboxSpan extends SpanBase {
362
- kind: 'sandbox';
363
- image?: string;
364
- command?: string;
365
- exitCode?: number;
366
- testsTotal?: number;
367
- testsPassed?: number;
368
- stdoutHash?: string;
369
- stderrHash?: string;
370
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
371
- wallMs?: number;
372
- }
373
- interface GenericSpan extends SpanBase {
374
- kind: 'agent' | 'custom';
375
- }
376
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
377
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
378
- interface TraceEvent {
379
- eventId: string;
380
- runId: string;
381
- spanId?: string;
382
- kind: EventKind;
383
- timestamp: number;
384
- payload: Record<string, unknown>;
385
- }
386
- interface BudgetLedgerEntry {
387
- runId: string;
388
- dimension: keyof BudgetSpec;
389
- limit: number;
390
- consumed: number;
391
- remaining: number;
392
- timestamp: number;
393
- breached: boolean;
394
- /** Span that triggered this entry, if any. */
395
- spanId?: string;
396
- }
397
- interface Artifact {
398
- artifactId: string;
399
- runId: string;
400
- spanId?: string;
401
- contentType: string;
402
- sizeBytes: number;
403
- /** sha256 in hex. */
404
- hash: string;
405
- /** External storage URL (R2, S3, filesystem path). */
406
- storageUrl?: string;
407
- /** Inline content for small blobs — keep under ~64KB. */
408
- inlineContent?: string;
409
- }
410
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
411
-
412
- /**
413
- * Paper-grade RunRecord schema + runtime validator.
414
- *
415
- * Every run that participates in a promotion gate, paper table, or
416
- * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
417
- * fields are exactly those the paper "Two Loops, Three Roles" requires
418
- * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
419
- * holdout split tag. A task score is optional because execution-only records
420
- * must preserve missing labels instead of converting errors into zero quality.
421
- *
422
- * This is intentionally NOT a replacement for the rich `Run` /
423
- * `ProposeReviewReport` / `ScenarioResult` types already in the
424
- * package. Those are runtime structures with full provenance. A
425
- * `RunRecord` is the analysis-time projection — the JSON-friendly
426
- * row you'd put in a parquet file or paste into a notebook.
427
- *
428
- * Validate at the boundary:
429
- *
430
- * const rec = validateRunRecord(rawJson) // throws on missing
431
- * const ok = isRunRecord(rawJson) // boolean check
432
- * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
433
- *
434
- * The validator runs in pure TS — zod is intentionally NOT a
435
- * dependency. Round-trip tested in `tests/run-record.test.ts`.
436
- */
437
-
438
- /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
439
- * combined train+test pool that the optimizer is allowed to read. */
440
- type RunSplitTag = 'search' | 'dev' | 'holdout';
441
- /**
442
- * Explicit execution-lifecycle result for a run.
443
- *
444
- * This is separate from task quality (`outcome`) and failure classification.
445
- * Producers set it only from root-run or process evidence.
446
- */
447
- type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown';
448
- interface RunTokenUsage {
449
- input: number;
450
- /** All generated tokens charged as output, including reasoning tokens. */
451
- output: number;
452
- /** Reasoning-token subset of `output`, when the provider reports it. */
453
- reasoning?: number;
454
- /** Prompt tokens served from a provider cache. */
455
- cached?: number;
456
- /** Prompt tokens written into a provider cache. */
457
- cacheWrite?: number;
458
- }
459
- /**
460
- * How a run's USD amount was obtained.
461
- */
462
- type RunCostProvenance = {
463
- kind: 'observed';
464
- usd: number;
465
- } | {
466
- kind: 'estimated';
467
- usd: number;
468
- } | {
469
- kind: 'uncaptured';
470
- usd: null;
471
- };
472
- interface RunJudgeMetadata {
473
- model: string;
474
- promptVersion: string;
475
- /** [0,1] confidence the judge declared. Constant judge confidence
476
- * across many runs is a fallback signal (see `canary.ts`). */
477
- confidence: number;
478
- /** True if the judge degraded to a fallback path (rules-only,
479
- * prior-call cache, etc.). The canary uses this to alert. */
480
- fallback: boolean;
481
- }
482
- /**
483
- * Per-judge / per-dimension breakdown for runs scored by an ensemble of
484
- * judges over a multi-dimensional rubric.
485
- *
486
- * The collapsed `outcome.searchScore` / `holdoutScore` carries the
487
- * composite the gate uses. The full breakdown belongs here so consumers
488
- * can answer "which judge disagreed?", "which dimension dragged the
489
- * composite down?", and "did half the panel fail?" without re-running.
490
- *
491
- * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
492
- * `composite` are convenience projections — derivable but precomputed so
493
- * downstream IRR primitives (`interRaterReliability`,
494
- * `corpusInterRaterAgreement`) and reporters don't pay the same
495
- * aggregation twice.
496
- *
497
- * Fail-loud discipline: judges that errored out land in `failedJudges`
498
- * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
499
- * run); the explicit list makes a partial-failure recorded as such.
500
- */
501
- interface JudgeScoresRecord {
502
- /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
503
- perJudge: Record<string, Record<string, number>>;
504
- /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
505
- perDimMean: Record<string, number>;
506
- /** Composite mean across successful judges. Mirrors the task score only
507
- * when `failedJudges` is empty. */
508
- composite: number;
509
- /** Judges that errored or returned an unparseable verdict. Recorded
510
- * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
511
- * not inferred from missing keys in `perJudge`. */
512
- failedJudges?: string[];
513
- /** Free-form notes the judges emitted (joined across judges or
514
- * first-judge only — consumer's choice). */
515
- notes?: string;
516
- }
517
- interface RunOutcome {
518
- /** Score on the search/optimization split. Optional for holdout-only and
519
- * execution-only records. */
520
- searchScore?: number;
521
- /** Score on the held-out split. Optional for search-only and execution-only
522
- * records. When both scores are absent, the run is explicitly unlabeled. */
523
- holdoutScore?: number;
524
- /** Bag of any other metric the run produced — judge dimensions,
525
- * pass/fail counters, latency stats, etc. Numeric only — keeps
526
- * reporters honest. */
527
- raw: Record<string, number>;
528
- /** Per-judge / per-dim breakdown. Consumers writing ensemble
529
- * judgements populate this; substrate primitives like
530
- * `interRaterReliability` and `corpusInterRaterAgreement` accept
531
- * these records as input. Optional — single-judge or scalar-only
532
- * runs leave it unset. */
533
- judgeScores?: JudgeScoresRecord;
534
- /** Authenticity / realness verdict — did the run build the REAL thing on the
535
- * intended infra, or fake it (see `./authenticity`)? Optional: only domains
536
- * with an authenticity config populate it. Carried in the corpus so the
537
- * flywheel / off-policy learning can optimize for real completion, not gamed
538
- * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
539
- * must not count as a real success regardless of `score`. */
540
- realness?: {
541
- score: number;
542
- gated: boolean;
543
- reason?: string;
544
- };
545
- }
546
- /**
547
- * Mandatory paper-grade fields for a single evaluation run. Optional
548
- * fields are extension points; mandatory fields throw if missing.
549
- *
550
- * Hash discipline:
551
- * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
552
- * model (after any steering bundle merge).
553
- * - `configHash` is the sha256 of the effective run config (model,
554
- * temperature, tools, judges, splits). The pair (promptHash,
555
- * configHash) uniquely identifies an experiment cell.
556
- *
557
- * Model snapshot discipline:
558
- * - `model` MUST encode a snapshot version. Bare aliases like
559
- * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
560
- * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
561
- */
562
- interface RunRecord {
563
- /** UUID for the run. */
564
- runId: string;
565
- /** Logical experiment grouping (a treatment vs a baseline within
566
- * the same sweep should share `experimentId`). */
567
- experimentId: string;
568
- /** Stable identifier for the candidate (variant) being run. The
569
- * promotion gate compares two `candidateId`s on matched items. */
570
- candidateId: string;
571
- /** RNG seed for the run. Always recorded — silent re-seeding is
572
- * the most common cause of non-reproducible numbers. */
573
- seed: number;
574
- /** Model identifier WITH snapshot version. */
575
- model: string;
576
- /** sha256 of the effective prompt (post-steering). */
577
- promptHash: string;
578
- /** sha256 of the effective config. */
579
- configHash: string;
580
- /** Git SHA the harness was run from. */
581
- commitSha: string;
582
- /** End-to-end wall-clock duration in milliseconds. */
583
- wallMs: number;
584
- /** Time spent queued before execution started, if known. */
585
- queueMs?: number;
586
- /** Total USD cost, or null when the producer could not capture one. */
587
- costUsd: number | null;
588
- /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */
589
- costProvenance: RunCostProvenance;
590
- /** Token usage breakdown. */
591
- tokenUsage: RunTokenUsage;
592
- /** Root-run or process terminal result. Never inferred from a child span. */
593
- terminalOutcome: RunTerminalOutcome;
594
- /** Root-run or process failure reason. Valid only for a failed, cancelled,
595
- * or incomplete terminal result; never populated from a child span. */
596
- terminalFailureReason?: string;
597
- /** Judge-side metadata, if a judge was used. */
598
- judgeMetadata?: RunJudgeMetadata;
599
- /** Per-split scores + raw bag. */
600
- outcome: RunOutcome;
601
- /** Canonical task-failure class drawn from the shared
602
- * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result
603
- * evidence. Execution errors belong in
604
- * `outcome.raw.execution_error_count`. */
605
- failureClass?: FailureClass;
606
- /** Free-form task-failure detail scoped under a non-success
607
- * `failureClass`. It is invalid without that class. */
608
- failureMode?: string;
609
- /** Which split this run was drawn from. */
610
- splitTag: RunSplitTag;
611
- /**
612
- * Stable scenario identifier the run observed or was scored against.
613
- * Comparison primitives match this identity rather than input order.
614
- */
615
- scenarioId: string;
616
- /**
617
- * Canonical identity for the agent profile cell that produced this row:
618
- * profile artifact hash plus optional harness/model/prompt/reporting
619
- * dimensions. Use `agentProfile.cellId` to group persona sweeps and
620
- * longitudinal reports by the complete source profile, not by a loose
621
- * candidate label or opaque config hash.
622
- */
623
- agentProfile?: AgentProfileCell;
624
- }
625
-
626
- /**
627
- * RawProviderSink — first-class persistence for the actual HTTP-level
628
- * request/response bodies of every LLM provider call.
629
- *
630
- * Why this is a separate sink from the structured `LlmSpan`:
631
- *
632
- * - `LlmSpan` records the *intent* — model name, messages, output text,
633
- * usage. It's what dashboards read; it's NOT enough for forensics.
634
- * - When a downstream consumer reports "the verifier used the wrong route"
635
- * or "tokens look right but reasoning was missing," the only way to
636
- * answer is the raw HTTP body. Span fields can lie (a proxy can echo
637
- * a different `model` value than what actually answered); the raw
638
- * response is ground truth.
639
- *
640
- * Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
641
- * matrix runner / BuilderSession sets it up automatically) and every
642
- * request, response, and error is recorded — including retries, with the
643
- * attempt index attached so a flaky call's full event chain is recoverable.
644
- *
645
- * Redaction is enforced at sink time. The default redactor strips
646
- * `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
647
- * payload field whose key matches `apiKey | api_key | bearer | password |
648
- * secret | token` (case-insensitive). Override via the sink constructor or
649
- * the per-call `redactor`. The `redactedFields` array on the persisted
650
- * event lets a reviewer see what was stripped without exposing the values.
651
- */
652
- type RawProviderDirection = 'request' | 'response' | 'error';
653
- interface RawProviderEvent {
654
- /** Stable id. Generated by the sink if omitted. */
655
- eventId: string;
656
- /** Trace context populated by `LlmClient` when the call is wrapped in a span. */
657
- runId?: string;
658
- spanId?: string;
659
- /**
660
- * Logical provider name. Free-form so callers can use whatever id matches
661
- * their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
662
- * omitted, derived from `baseUrl` in `LlmClientOptions`.
663
- */
664
- provider: string;
665
- model: string;
666
- /** Endpoint path, e.g. `'/v1/chat/completions'`. */
667
- endpoint: string;
668
- /** Base URL used for the call (already-normalised — no trailing slash). */
669
- baseUrl: string;
670
- /** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
671
- attemptIndex: number;
672
- direction: RawProviderDirection;
673
- /** Unix ms. */
674
- timestamp: number;
675
- /** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
676
- durationMs?: number;
677
- statusCode?: number;
678
- requestHeaders?: Record<string, string>;
679
- requestBody?: unknown;
680
- responseHeaders?: Record<string, string>;
681
- responseBody?: unknown;
682
- /** Set on `direction: 'error'` events. */
683
- errorMessage?: string;
684
- /** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
685
- redactedFields: string[];
686
- }
687
- interface RawProviderSinkFilter {
688
- runId?: string;
689
- spanId?: string;
690
- direction?: RawProviderDirection;
691
- attemptIndex?: number;
692
- }
693
- interface RawProviderSink {
694
- record(event: RawProviderEvent): Promise<void>;
695
- /** Optional listing — implementations that durably persist (file, db) should support this. */
696
- list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
697
- /** Optional teardown for backed implementations. */
698
- close?(): Promise<void>;
699
- }
700
- type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
701
-
702
- interface RunFilter {
703
- scenarioId?: string;
704
- variantId?: string;
705
- status?: RunStatus;
706
- since?: number;
707
- until?: number;
708
- tag?: {
709
- key: string;
710
- value: string;
711
- };
712
- parentRunId?: string;
713
- projectId?: string;
714
- chatId?: string;
715
- layer?: RunLayer;
716
- }
717
- interface SpanFilter {
718
- runId?: string;
719
- parentSpanId?: string;
720
- kind?: SpanKind;
721
- name?: string;
722
- toolName?: string;
723
- judgeId?: string;
724
- since?: number;
725
- until?: number;
726
- }
727
- interface EventFilter {
728
- runId?: string;
729
- spanId?: string;
730
- kind?: EventKind;
731
- since?: number;
732
- until?: number;
733
- }
734
- interface TraceStore {
735
- appendRun(run: Run): Promise<void>;
736
- updateRun(runId: string, patch: Partial<Run>): Promise<void>;
737
- appendSpan(span: Span): Promise<void>;
738
- updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
739
- appendEvent(event: TraceEvent): Promise<void>;
740
- appendArtifact(artifact: Artifact): Promise<void>;
741
- appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
742
- getRun(runId: string): Promise<Run | undefined>;
743
- listRuns(filter?: RunFilter): Promise<Run[]>;
744
- spans(filter?: SpanFilter): Promise<Span[]>;
745
- events(filter?: EventFilter): Promise<TraceEvent[]>;
746
- budget(runId: string): Promise<BudgetLedgerEntry[]>;
747
- artifacts(runId: string): Promise<Artifact[]>;
748
- }
749
-
750
- interface RunTrace {
751
- run: Run;
752
- spans: Span[];
753
- events: TraceEvent[];
754
- artifacts: Artifact[];
755
- budget: BudgetLedgerEntry[];
756
- }
757
- interface RunCriticOptions {
758
- weights?: Partial<RunScoreWeights>;
759
- driftPatterns?: RegExp[];
760
- }
761
- declare class RunCritic {
762
- private readonly weights?;
763
- private readonly driftPatterns;
764
- constructor(options?: RunCriticOptions);
765
- score(store: TraceStore, runId: string): Promise<RunScore>;
766
- scoreTrace(trace: RunTrace): RunScore;
767
- rank(score: RunScore): number;
768
- private isDrift;
769
- }
770
-
771
- type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
772
- interface CostUsage {
773
- inputTokens: number;
774
- /** Includes reasoning tokens when the provider bills them as output. */
775
- outputTokens: number;
776
- /** Reasoning-token subset of outputTokens, when reported. */
777
- reasoningTokens?: number;
778
- /** Prompt tokens served from a provider cache. */
779
- cachedTokens?: number;
780
- /** Prompt tokens written into a provider cache. */
781
- cacheWriteTokens?: number;
782
- }
783
- interface CostCallBase {
784
- callId: string;
785
- channel: CostChannel;
786
- phase: string;
787
- actor: string;
788
- model: string;
789
- maximumCostUsd?: number;
790
- tags?: Record<string, string>;
791
- timestamp: number;
792
- }
793
- interface PendingCostCall extends CostCallBase {
794
- status: 'pending';
795
- }
796
- interface PendingCostCallView extends PendingCostCall {
797
- state: 'active' | 'late' | 'interrupted';
798
- }
799
- interface CostReceipt extends CostCallBase, CostUsage {
800
- status: 'settled';
801
- costUsd: number;
802
- costUnknown: boolean;
803
- usageUnknown?: boolean;
804
- /** Rates used to estimate cost locally. Absent when cost is provider-reported or unknown. */
805
- pricing?: {
806
- inputUsdPerThousand: number;
807
- cachedInputUsdPerThousand?: number;
808
- cacheWriteUsdPerThousand?: number;
809
- outputUsdPerThousand: number;
810
- };
811
- /** Cost reported by the provider, not a local token-price calculation. */
812
- actualCostUsd?: number;
813
- error?: string;
814
- }
815
- interface CostReceiptInput extends CostUsage {
816
- model: string;
817
- /** Caller-supplied rates for a local estimate when the provider does not report billed cost. */
818
- customTokenPricing?: CustomTokenPricing;
819
- actualCostUsd?: number;
820
- costUnknown?: boolean;
821
- usageUnknown?: boolean;
822
- }
823
- /** Per-million token rates for a model or endpoint not covered by package pricing. */
824
- interface CustomTokenPricing {
825
- /** Non-cached input tokens. */
826
- inputUsdPerMillion: number;
827
- /** Cache-read tokens. Falls back to the normal input rate when omitted. */
828
- cachedInputUsdPerMillion?: number;
829
- /** Cache-creation or cache-write tokens. Falls back to the normal input rate when omitted. */
830
- cacheWriteUsdPerMillion?: number;
831
- outputUsdPerMillion: number;
832
- }
833
- type MaximumCharge = {
834
- externallyEnforcedMaximumUsd: number;
835
- } | ({
836
- customTokenPricing: CustomTokenPricing;
837
- } & Pick<CostUsage, 'inputTokens' | 'outputTokens' | 'cachedTokens' | 'cacheWriteTokens'>) | ({
838
- model: string;
839
- } & CostUsage);
840
- interface RunPaidCallInput<T> {
841
- callId?: string;
842
- channel: CostChannel;
843
- phase: string;
844
- actor: string;
845
- /** Used before a provider receipt exists and on failures without one. */
846
- model?: string;
847
- tags?: Record<string, string>;
848
- signal?: AbortSignal;
849
- /** Provider-enforced dollar maximum, or maximum token usage with known pricing. Required when capped. */
850
- maximumCharge?: MaximumCharge;
851
- /** `callId` can be forwarded as the provider's idempotency key. */
852
- execute(signal: AbortSignal, callId: string): Promise<T>;
853
- receipt(value: T): CostReceiptInput;
854
- receiptFromError?(error: Error): CostReceiptInput | undefined;
855
- }
856
- type PaidCallResult<T> = {
857
- succeeded: true;
858
- callId: string;
859
- value: T;
860
- receipt: CostReceipt;
861
- } | {
862
- succeeded: false;
863
- callId?: string;
864
- error: Error;
865
- receipt?: CostReceipt;
866
- };
867
- interface ChannelRollup {
868
- channel: CostChannel;
869
- calls: number;
870
- inputTokens: number;
871
- outputTokens: number;
872
- reasoningTokens?: number;
873
- cachedTokens: number;
874
- cacheWriteTokens?: number;
875
- costUsd: number;
876
- unpricedCalls: number;
877
- unknownUsageCalls: number;
878
- }
879
- interface CostLedgerSummary {
880
- totalCalls: number;
881
- pendingCalls: number;
882
- unresolvedCalls: number;
883
- reservedCostUsd: number;
884
- inputTokens: number;
885
- outputTokens: number;
886
- reasoningTokens?: number;
887
- cachedTokens: number;
888
- cacheWriteTokens?: number;
889
- totalCostUsd: number;
890
- byChannel: ChannelRollup[];
891
- unpricedModels: string[];
892
- fullyPriced: boolean;
893
- usageComplete: boolean;
894
- accountingComplete: boolean;
895
- incompleteReasons: string[];
896
- }
897
- interface CostLedgerFilter {
898
- channel?: CostChannel;
899
- phase?: string;
900
- tags?: Record<string, string>;
901
- }
902
- interface CostLedgerWaitOptions {
903
- /** Maximum time to wait for active provider calls. Default 5 seconds. */
904
- timeoutMs?: number;
905
- /** Wait only for calls matching this attribution filter. */
906
- filter?: CostLedgerFilter;
907
- }
908
- /** Append-only storage. `append` must atomically reject stale revisions. */
909
- interface CostLedgerPersistence {
910
- read(): {
911
- revision: string;
912
- events: string;
913
- };
914
- append(expectedRevision: string, event: string): string | undefined;
915
- }
916
- interface CostLedgerOptions {
917
- costCeilingUsd?: number;
918
- persistence?: CostLedgerPersistence;
919
- /** Import already-settled receipts without admitting new paid work. */
920
- receipts?: readonly CostReceipt[];
921
- }
922
- /** Run-wide paid-call admission, durable call state, receipts, and summaries. */
923
- declare class CostLedger {
924
- private readonly records;
925
- private readonly activeCallIds;
926
- private readonly lateCallIds;
927
- private readonly idleWaiters;
928
- private completedTasks;
929
- private revision;
930
- private costLimitPersisted;
931
- readonly costCeilingUsd?: number;
932
- private readonly persistence?;
933
- constructor(input?: number | CostLedgerOptions);
934
- runPaidCall<T>(input: RunPaidCallInput<T>): Promise<PaidCallResult<T>>;
935
- /** Wait until every call started by this ledger has produced a durable outcome. */
936
- waitForIdle(options?: CostLedgerWaitOptions): Promise<boolean>;
937
- /** Settle a call left pending by a crashed process after reconciling with the provider. */
938
- reconcile(callId: string, observed: CostReceiptInput, options?: {
939
- error?: string;
940
- }): CostReceipt;
941
- list(filter?: CostLedgerFilter): CostReceipt[];
942
- /** Read pending calls without exposing mutable ledger state. */
943
- listPending(filter?: CostLedgerFilter): PendingCostCallView[];
944
- summary(filter?: CostLedgerFilter): CostLedgerSummary;
945
- markCompleted(count?: number): void;
946
- costPerCompletedTask(): number | null;
947
- private execute;
948
- private captureLateOutcome;
949
- private releaseActiveCall;
950
- private commitOutcome;
951
- private captureFailure;
952
- private commitReceipt;
953
- private resolveMaximum;
954
- private hasIncompleteSettledCall;
955
- private appendRecord;
956
- private ensureCostLimitPersisted;
957
- private appendEvent;
958
- }
959
- /** Public callback surface for a shared cost ledger.
960
- *
961
- * Declaration bundles may expose this type through multiple package subpaths.
962
- * Keeping callback contracts structural lets those subpaths compose while the
963
- * concrete {@link CostLedger} retains its private durable state.
964
- */
965
- type CostLedgerHandle = Pick<CostLedger, Exclude<keyof CostLedger, 'listPending' | 'waitForIdle'>> & Partial<Pick<CostLedger, 'listPending' | 'waitForIdle'>>;
966
-
967
- /**
968
- * LLM client with graceful degrade.
969
- *
970
- * OpenAI-compatible `/v1/chat/completions` client with:
971
- * - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
972
- * - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
973
- * - One retry at temperature 1 when a model explicitly requires it.
974
- * - Graceful json_schema → json_object degrade on 400 with schema-reject body.
975
- * - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
976
- * - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
977
- * directly, cli-bridge subscriptions, and any router that speaks the spec.
978
- *
979
- * Usage:
980
- * const { value, result } = await callLlmJson<MyType>(
981
- * { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
982
- * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
983
- * )
984
- *
985
- * `createChatClient` wraps this implementation for provider-neutral package
986
- * entry points. Direct callers can use `callLlm` or `callLlmJson`.
987
- */
988
-
989
- interface LlmMessage {
990
- role: 'system' | 'user' | 'assistant';
991
- /**
992
- * Either a plain text content string OR a multimodal content array
993
- * (text + image_url parts) for vision-capable models.
994
- */
995
- content: string | Array<{
996
- type: 'text';
997
- text: string;
998
- } | {
999
- type: 'image_url';
1000
- image_url: {
1001
- url: string;
1002
- detail?: 'auto' | 'low' | 'high';
1003
- };
1004
- }>;
1005
- }
1006
- type LlmThinkingMode = 'enabled' | 'disabled';
1007
- interface LlmCallRequest {
1008
- model: string;
1009
- messages: LlmMessage[];
1010
- /** Optional JSON-mode response format (response_format: json_object). */
1011
- jsonMode?: boolean;
1012
- /** Optional structured output via JSON Schema. Falls back to json_object on 400. */
1013
- jsonSchema?: {
1014
- name: string;
1015
- schema: Record<string, unknown>;
1016
- };
1017
- temperature?: number;
1018
- maxTokens?: number;
1019
- /** OpenAI-compatible reasoning mode. Omitted when the provider default should apply. */
1020
- thinking?: LlmThinkingMode;
1021
- /** Per-call timeout, default 300s. */
1022
- timeoutMs?: number;
1023
- }
1024
- interface LlmUsage {
1025
- promptTokens: number;
1026
- completionTokens: number;
1027
- totalTokens: number;
1028
- /** False when the provider omitted or malformed prompt/completion usage. */
1029
- captured?: boolean;
1030
- /** Reasoning-token subset of completionTokens, when reported. */
1031
- reasoningTokens?: number;
1032
- /** Proxies populate this when prompt caching is on. */
1033
- cachedPromptTokens?: number;
1034
- }
1035
- interface LlmCallResult {
1036
- /** The text content of the first choice. Empty string if none. */
1037
- content: string;
1038
- usage: LlmUsage;
1039
- /**
1040
- * Cost in USD. Uses the provider's reported cost when present, otherwise
1041
- * caller-supplied token pricing. `null` when neither is available.
1042
- */
1043
- costUsd: number | null;
1044
- /** Model name actually used (echoed from response). */
1045
- model: string;
1046
- /** Wall-clock duration of the HTTP call (last attempt, if retried). */
1047
- durationMs: number;
1048
- /**
1049
- * `finish_reason` echoed from the first choice (`stop`, `length`,
1050
- * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
1051
- * Exposed so a free-form `callLlm` caller CAN detect a truncated answer
1052
- * (`length`) instead of treating a cut-off completion as complete. Note:
1053
- * `callLlm` does not itself reject on it — acting on this signal is the
1054
- * caller's responsibility (in-repo free-form drivers do not yet enforce it).
1055
- */
1056
- finishReason?: string | null;
1057
- /**
1058
- * True when `content.trim()` is empty. An empty completion is a silent zero
1059
- * for free-form `callLlm` callers; this flag is the signal a caller can
1060
- * inspect to fail loud rather than proceed on an empty string. `callLlm`
1061
- * surfaces it but does not throw on it.
1062
- */
1063
- contentEmpty?: boolean;
1064
- /** Raw response body. */
1065
- raw: Record<string, unknown>;
1066
- }
1067
- interface LlmClientOptions {
1068
- /** Base URL (without trailing slash). Must end at the `/v1` prefix. */
1069
- baseUrl?: string;
1070
- /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
1071
- apiKey?: string;
1072
- bearer?: string;
1073
- /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
1074
- authHeader?: {
1075
- name: string;
1076
- value: string;
1077
- };
1078
- /** Stable provider idempotency key, reused across retries of this logical call. */
1079
- idempotencyKey?: string;
1080
- /** Default timeout in ms. Per-call can override. */
1081
- defaultTimeoutMs?: number;
1082
- /**
1083
- * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
1084
- * each attempt's per-attempt timeout controller, so aborting it cancels
1085
- * the in-flight fetch. A caller abort is FATAL: it is not retried even
1086
- * though an AbortError otherwise matches the transient patterns.
1087
- */
1088
- signal?: AbortSignal;
1089
- /**
1090
- * Cross-attempt wall-clock budget in ms, measured from the first attempt.
1091
- * Before launching each attempt the loop checks the remaining budget and
1092
- * stops retrying once it is exhausted, rather than waiting the full
1093
- * per-attempt timeout on every retry. Bounds total time independent of
1094
- * total attempts × `timeoutMs`.
1095
- */
1096
- deadlineMs?: number;
1097
- /** Total provider attempts. Default 3. */
1098
- maximumAttempts?: number;
1099
- /** Token rates used when the provider omits cost or package pricing does not cover the model. */
1100
- customTokenPricing?: CustomTokenPricing;
1101
- /**
1102
- * Transport for requests that declare `jsonSchema`. `native` sends
1103
- * `response_format: json_schema`; `json-object` sends the broadly supported
1104
- * JSON mode and relies on the caller to include the schema in model-visible
1105
- * instructions. Default: `native`.
1106
- */
1107
- jsonSchemaTransport?: 'native' | 'json-object';
1108
- /**
1109
- * JSON payload parsing policy. `extract` accepts fenced or prose-prefixed JSON.
1110
- * `exact` requires the complete response content to be one JSON value.
1111
- * Default: `extract`.
1112
- */
1113
- jsonPayloadMode?: 'extract' | 'exact';
1114
- /** Default provider reasoning mode. A per-call request value takes precedence. */
1115
- thinking?: LlmThinkingMode;
1116
- /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
1117
- fetch?: typeof fetch;
1118
- /**
1119
- * Optional raw HTTP capture sink. When provided, every request, response,
1120
- * and error (across all retry attempts) is recorded to the sink, with auth
1121
- * headers and credential-shaped body fields redacted by default. This is
1122
- * the layer-1 forensics primitive: structured `LlmSpan`s record intent,
1123
- * raw events record what actually crossed the wire.
1124
- */
1125
- rawSink?: RawProviderSink;
1126
- /**
1127
- * Logical provider id attached to raw events. When omitted, derived from
1128
- * `baseUrl` via `providerFromBaseUrl`.
1129
- */
1130
- provider?: string;
1131
- /** Trace context attached to raw events; populated by emitter-aware callers. */
1132
- traceContext?: {
1133
- runId?: string;
1134
- spanId?: string;
1135
- };
1136
- /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
1137
- redactor?: ProviderRedactor;
1138
- }
1139
-
1140
- /**
1141
- * Semantic concept judge — "does the built artifact actually implement
1142
- * the features the user asked for?"
1143
- *
1144
- * Distinct from the domain/code/coherence judges in `judges.ts`:
1145
- * - those judges score free-form conversational agent outputs along
1146
- * quality dimensions (accuracy, depth, etc.)
1147
- * - this judge scores a *built artifact* (served HTML + source files)
1148
- * against an explicit list of expected concepts, returning per-concept
1149
- * {present, score 0-10, evidence, severity}.
1150
- *
1151
- * The judge is strict about distinguishing (a) a working implementation
1152
- * from (b) a keyword-present stub. "// TODO: mint button" is NOT present.
1153
- * Only real, functional, wired-up code counts.
1154
- *
1155
- * Use via {@link createSemanticConceptJudge} or directly via
1156
- * {@link runSemanticConceptJudge}. Soft-fails (available=false) on LLM
1157
- * or JSON-parse errors so the caller can treat that as "layer skipped"
1158
- * rather than "layer failed" in a multi-layer pipeline.
1159
- */
1160
-
1161
- /**
1162
- * Implementation complexity class for weighted scoring.
1163
- *
1164
- * - `render` (default): the concept is a UI surface that displays static
1165
- * data — render a list, show a counter, lay out a button. Single-file
1166
- * work, no external integration.
1167
- * - `integrate`: the concept requires wiring a real external system —
1168
- * wallet connect (wagmi + RainbowKit + chain config), payment provider
1169
- * (Stripe Elements + intent + webhook), an API client with auth.
1170
- * Multi-file, library-knowledge, runtime correctness matters.
1171
- * - `compute`: the concept requires algorithmic work — solver, simulator,
1172
- * constraint propagation, ML inference. Correctness > UI polish.
1173
- *
1174
- * Default weights (when applied via `weightConcepts: 'complexity'`):
1175
- * render=1.0, integrate=2.0, compute=2.5
1176
- *
1177
- * Cross-vertical scoring without complexity weighting silently inflates
1178
- * the rate of UI-heavy verticals (healthcare, fintech dashboards) vs
1179
- * integration-heavy verticals (DeFi, wallets) — all concepts treated
1180
- * equally even though the agent does 2-3x the work for `integrate`.
1181
- */
1182
- type ConceptComplexity = 'render' | 'integrate' | 'compute';
1183
- interface ConceptSpec {
1184
- name: string;
1185
- /** Short hints that help the judge; not used for matching. */
1186
- keywords?: string[];
1187
- /** Optional explicit weight; default 1.0. Overrides complexity-derived weight. */
1188
- weight?: number;
1189
- /** Implementation complexity class. Default `render`. */
1190
- complexity?: ConceptComplexity;
1191
- }
1192
- interface SemanticConceptJudgeInput {
1193
- /** Full natural-language prompt the agent was handed. */
1194
- userRequest: string;
1195
- /** Rendered HTML the preview returns (UI artifacts). Optional. */
1196
- servedHtml?: string;
1197
- /** Top-level source files from the agent's workdir. */
1198
- sourceFiles: Array<{
1199
- path: string;
1200
- content: string;
1201
- }>;
1202
- /** The expected concept list. */
1203
- expectedConcepts: ConceptSpec[];
1204
- /** Free-form metadata (id, difficulty) to inject into the prompt. */
1205
- artifactLabel?: string;
1206
- artifactDescription?: string;
1207
- }
1208
- /**
1209
- * Score-aggregation strategy. `mean` averages 0-10 scores uniformly.
1210
- * `complexity` applies the default weight table (render=1, integrate=2,
1211
- * compute=2.5) unless a concept has an explicit `weight`. `explicit`
1212
- * honors only `weight` (defaulting to 1 for unspecified).
1213
- */
1214
- type ConceptWeightStrategy = 'mean' | 'complexity' | 'explicit';
1215
- interface SemanticConceptJudgeOptions {
1216
- /** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */
1217
- model?: string;
1218
- /** Per-call timeout. Default 300s. */
1219
- timeoutMs?: number;
1220
- /** Provider-enforced output limit. Default 16000. */
1221
- maxTokens?: number;
1222
- /** Pipeline budget for the prompt (source blob truncation). Default 45000. */
1223
- maxSourceChars?: number;
1224
- /** Per-file cap before inclusion. Default 20000. */
1225
- maxPerFileChars?: number;
1226
- /** HTML cap. Default 30000. */
1227
- maxHtmlChars?: number;
1228
- /** LlmClient config (baseUrl, apiKey, authHeader, …). */
1229
- llm?: LlmClientOptions;
1230
- costLedger?: CostLedgerHandle;
1231
- costPhase?: string;
1232
- costTags?: Record<string, string>;
1233
- signal?: AbortSignal;
1234
- /**
1235
- * Score aggregation strategy. Default `mean` — uniform average across
1236
- * concepts. Cross-vertical comparisons should use `complexity` to
1237
- * neutralize the integrate-vs-render asymmetry.
1238
- */
1239
- weightConcepts?: ConceptWeightStrategy;
1240
- /** Override the default complexity → weight table. */
1241
- complexityWeights?: Partial<Record<ConceptComplexity, number>>;
1242
- }
1243
-
1244
- /**
1245
- * Provider-neutral chat contract for every model call made by agent-eval.
1246
- *
1247
- * Callers choose the transport at the package boundary with `createChatClient`.
1248
- * Evaluation code receives canonical requests and results without importing a
1249
- * provider SDK.
1250
- */
1251
-
1252
- /**
1253
- * Unified chat interface using the package's canonical LLM request and result.
1254
- */
1255
- interface ChatClient {
1256
- /** Display name of the bound transport, included in telemetry. */
1257
- readonly transport: ChatTransport;
1258
- /** Default model when the caller omits one. */
1259
- readonly defaultModel?: string;
1260
- /** Total provider attempts this transport can make for one chat call. */
1261
- readonly maximumAttempts?: number;
1262
- /** Implementations must enforce `req.maxTokens` when it is present. */
1263
- chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
1264
- }
1265
- type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'custom' | 'mock';
1266
- interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
1267
- /** Optional — falls back to ChatClient.defaultModel. */
1268
- model?: string;
1269
- }
1270
- type ChatResponse = LlmCallResult;
1271
- interface ChatCallOpts {
1272
- /** Cancel the in-flight request. */
1273
- signal?: AbortSignal;
1274
- /** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
1275
- maxCostUsd?: number;
1276
- /** Correlation tag carried into request headers when the transport allows. */
1277
- correlationId?: string;
1278
- /** Stable provider idempotency key for retries/redrives of one paid call. */
1279
- idempotencyKey?: string;
1280
- }
1281
- type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | CustomTransportOpts | MockTransportOpts;
1282
- interface BaseTransportOpts {
1283
- defaultModel?: string;
1284
- /** Total provider attempts. Required for opaque transports used in capped runs. */
1285
- maximumAttempts?: number;
1286
- }
1287
- interface RouterTransportOpts extends BaseTransportOpts {
1288
- transport: 'router';
1289
- baseUrl?: string;
1290
- apiKey: string;
1291
- }
1292
- interface CliBridgeTransportOpts extends BaseTransportOpts {
1293
- transport: 'cli-bridge';
1294
- baseUrl?: string;
1295
- bearer?: string;
1296
- }
1297
- interface DirectProviderTransportOpts extends BaseTransportOpts {
1298
- transport: 'direct-provider';
1299
- baseUrl: string;
1300
- apiKey: string;
1301
- }
1302
- /**
1303
- * Sandbox-SDK transport. The caller supplies a canonical chat function for an
1304
- * already-configured Sandbox handle, so agent-eval does not import the SDK.
1305
- */
1306
- interface SandboxSdkTransportOpts extends BaseTransportOpts {
1307
- transport: 'sandbox-sdk';
1308
- chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
1309
- }
1310
- /** Caller-adapted SDK or transport returning the canonical ChatResponse shape. */
1311
- interface CustomTransportOpts extends BaseTransportOpts {
1312
- transport: 'custom';
1313
- chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
1314
- }
1315
- /**
1316
- * Mock transport for tests. The handler receives the request and returns
1317
- * whatever the test wants. No retries, no JSON-schema degrade.
1318
- */
1319
- interface MockTransportOpts extends BaseTransportOpts {
1320
- transport: 'mock';
1321
- handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
1322
- }
1323
- /**
1324
- * Build a ChatClient bound to a specific transport. The returned client
1325
- * is safe to share across analysts in a single registry run.
1326
- */
1327
- declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
1328
-
1329
- interface Scenario {
1330
- id: string;
1331
- persona: string;
1332
- label: string;
1333
- thesis: string;
1334
- dimensions: string[];
1335
- turns: Turn[];
1336
- artifactChecks: ArtifactCheck[];
1337
- systemPromptAppend?: string;
1338
- }
1339
- interface Turn {
1340
- user: string;
1341
- expectedBehaviors: string[];
1342
- adversarial?: boolean;
1343
- feedbackType?: 'correction' | 'rejection' | 'vague' | 'contradictory' | 'escalation';
1344
- }
1345
- interface ArtifactCheck {
1346
- type: 'vault_file_exists' | 'vault_file_contains' | 'block_extracted' | 'code_valid' | 'generation_produced' | 'tool_created' | string;
1347
- target: string;
1348
- contains?: string;
1349
- minCount?: number;
1350
- description: string;
1351
- }
1352
- interface TurnResult {
1353
- turnIndex: number;
1354
- userMessage: string;
1355
- agentResponse: string;
1356
- durationMs: number;
1357
- blocksExtracted: {
1358
- type: string;
1359
- title: string;
1360
- }[];
1361
- containsCode: boolean;
1362
- containsToolCall: boolean;
1363
- }
1364
- interface JudgeScore {
1365
- judgeName: string;
1366
- dimension: string;
1367
- score: number;
1368
- reasoning: string;
1369
- evidence?: string;
1370
- }
1371
- interface CollectedArtifacts {
1372
- vaultFiles: {
1373
- path: string;
1374
- content: string;
1375
- }[];
1376
- blocksExtracted: {
1377
- type: string;
1378
- fields: Record<string, string>;
1379
- }[];
1380
- codeBlocks: {
1381
- language: string;
1382
- code: string;
1383
- }[];
1384
- toolCalls: string[];
1385
- }
1386
- interface JudgeInput {
1387
- scenario: Scenario;
1388
- turns: TurnResult[];
1389
- artifacts: CollectedArtifacts;
1390
- /** Shared ledger for paid built-in judges. Direct calls default to an uncapped ledger. */
1391
- costLedger?: CostLedgerHandle;
1392
- costPhase?: string;
1393
- costTags?: Record<string, string>;
1394
- signal?: AbortSignal;
1395
- }
1396
- type JudgeFn = (chat: ChatClient, input: JudgeInput) => Promise<JudgeScore[]>;
1397
-
1398
- /**
1399
- * Shared types for the trace-analyst module.
1400
- *
1401
- * Wire format. The store interface speaks `OtlpSpanLike` rows — one JSONL
1402
- * line per span, OTLP-shaped. We do NOT depend on a specific tracing
1403
- * vendor at the type level. Adapter
1404
- * layers map upstream shapes onto this interface.
1405
- *
1406
- * Design constraint. Every read operation that can return arbitrary
1407
- * payload must carry a byte budget so the agent's tool result stays
1408
- * bounded regardless of input trace size. Oversized responses
1409
- * substitute a deterministic summary instead of bytes — see
1410
- * `ViewTraceOversized`.
1411
- */
1412
- /** OTLP span kind (subset we actually use). */
1413
- type TraceAnalystSpanKind = 'AGENT' | 'LLM' | 'TOOL' | 'CHAIN' | 'EVALUATOR' | 'GUARDRAIL' | 'SPAN' | 'UNKNOWN';
1414
- type TraceAnalystSpanStatus = 'OK' | 'ERROR' | 'UNSET';
1415
- /** Subset of OTLP span fields the analyst exposes to the agent. The
1416
- * store's job is to project upstream's full span shape down to this
1417
- * view — the analyst never sees vendor extensions directly. */
1418
- interface TraceAnalystSpan {
1419
- trace_id: string;
1420
- span_id: string;
1421
- parent_span_id: string | null;
1422
- name: string;
1423
- kind: TraceAnalystSpanKind;
1424
- start_time: string;
1425
- end_time: string;
1426
- duration_ms: number;
1427
- status: TraceAnalystSpanStatus;
1428
- status_message?: string;
1429
- service_name: string | null;
1430
- agent_name: string | null;
1431
- model_name: string | null;
1432
- tool_name: string | null;
1433
- /** Raw JSON-serialisable attribute map. May contain large strings;
1434
- * callers must respect the per-attribute byte cap. */
1435
- attributes: Record<string, unknown>;
1436
- }
1437
- interface TraceAnalystTraceSummary {
1438
- trace_id: string;
1439
- service_name: string | null;
1440
- agent_name: string | null;
1441
- span_count: number;
1442
- has_errors: boolean;
1443
- start_time: string;
1444
- end_time: string;
1445
- duration_ms: number;
1446
- raw_jsonl_bytes: number;
1447
- models: string[];
1448
- tools: string[];
1449
- }
1450
- interface TraceAnalystFilters {
1451
- /** Restrict to traces that contain at least one error span. */
1452
- has_errors?: boolean;
1453
- /** Match if any span's `service.name` is in this list. */
1454
- service_names?: string[];
1455
- /** Match if any span's `agent.name` is in this list. */
1456
- agent_names?: string[];
1457
- /** Match if any LLM span's `llm.model_name` is in this list. */
1458
- model_names?: string[];
1459
- /** Match if any tool span's `tool.name` is in this list. */
1460
- tool_names?: string[];
1461
- /** ISO-8601 lower bound on the trace's earliest start time. */
1462
- start_time_after?: string;
1463
- /** ISO-8601 upper bound on the trace's earliest start time. */
1464
- start_time_before?: string;
1465
- /** Single regex applied to raw JSONL bytes for the trace. Opt-in;
1466
- * expensive on large datasets. Use the indexed filters above first. */
1467
- regex_pattern?: string;
1468
- }
1469
- /** One distinct error signature across the dataset — the deterministic unit of
1470
- * failure coverage. Signatures normalize volatile tokens (digits, hex/uuids,
1471
- * paths, durations) out of the span `status_message` so semantically identical
1472
- * failures collapse into one cluster. An analyst that accounts for every
1473
- * cluster has, by construction, covered every distinct failure mode. */
1474
- interface ErrorCluster {
1475
- /** Normalized status_message — the cluster key. */
1476
- signature: string;
1477
- /** A verbatim, un-normalized exemplar message (for exact-string citation). */
1478
- status_message_sample: string;
1479
- /** The span name that most often carries this signature, if any. */
1480
- span_name: string | null;
1481
- /** The tool that most often carries this signature, if any. */
1482
- tool_name: string | null;
1483
- trace_count: number;
1484
- span_count: number;
1485
- /** trace_count / total error traces in the matched set (0..1). */
1486
- prevalence: number;
1487
- /** Real trace ids carrying this signature (capped), passable to view/search. */
1488
- exemplar_trace_ids: string[];
1489
- /** Real span ids carrying this signature (capped). */
1490
- exemplar_span_ids: string[];
1491
- }
1492
- interface DatasetOverview {
1493
- total_traces: number;
1494
- raw_jsonl_bytes: number;
1495
- services: string[];
1496
- agents: string[];
1497
- models: string[];
1498
- tool_names: string[];
1499
- /** Up to 20 real trace ids the agent may pass to view/search tools. */
1500
- sample_trace_ids: string[];
1501
- errors: {
1502
- trace_count: number;
1503
- span_count: number;
1504
- };
1505
- /** The COMPLETE deterministic error-signature population, sorted by
1506
- * trace_count desc. This is the failure-coverage checklist: an analysis is
1507
- * complete only when every cluster here is accounted for. Empty when the
1508
- * matched set has no error spans. */
1509
- error_clusters: ErrorCluster[];
1510
- time_range: {
1511
- earliest: string;
1512
- latest: string;
1513
- } | null;
1514
- }
1515
- interface QueryTracesPage {
1516
- traces: TraceAnalystTraceSummary[];
1517
- total: number;
1518
- has_more: boolean;
1519
- }
1520
- /** Full-trace view. When the response would exceed the per-call byte
1521
- * budget, `oversized` is populated INSTEAD of `spans` so the agent
1522
- * knows to switch to `searchTrace` / `viewSpans`. */
1523
- interface ViewTraceResult {
1524
- trace_id: string;
1525
- spans?: TraceAnalystSpan[];
1526
- oversized?: ViewTraceOversized;
1527
- }
1528
- interface ViewTraceOversized {
1529
- span_count: number;
1530
- /** Names with their counts, sorted desc. Capped at 20 entries. */
1531
- top_span_names: Array<[string, number]>;
1532
- /** Largest single span body (bytes after attribute-cap projection). */
1533
- span_response_bytes_max: number;
1534
- error_span_count: number;
1535
- }
1536
- interface ViewSpansResult {
1537
- trace_id: string;
1538
- spans: TraceAnalystSpan[];
1539
- /** Number of requested span ids that were not found in the trace. */
1540
- missing_span_ids: string[];
1541
- /** Number of attribute fields truncated to fit the per-attribute cap. */
1542
- truncated_attribute_count: number;
1543
- }
1544
- interface SpanMatchRecord {
1545
- trace_id: string;
1546
- span_id: string;
1547
- span_name: string;
1548
- span_kind: TraceAnalystSpanKind;
1549
- /** JSON pointer-style path to the matched value, e.g.
1550
- * `attributes."llm.input_messages"[2].content`. */
1551
- attribute_path: string;
1552
- matched_text: string;
1553
- context_before: string;
1554
- context_after: string;
1555
- match_offset: number;
1556
- }
1557
- interface SearchTraceResult {
1558
- trace_id: string;
1559
- hits: SpanMatchRecord[];
1560
- total_matches: number;
1561
- has_more: boolean;
1562
- }
1563
- interface SearchSpanResult {
1564
- trace_id: string;
1565
- span_id: string;
1566
- hits: SpanMatchRecord[];
1567
- total_matches: number;
1568
- has_more: boolean;
1569
- }
1570
-
1571
- /**
1572
- * `TraceAnalysisStore` — read-side interface the trace-analyst calls
1573
- * through. Six operations, all bounded:
1574
- *
1575
- * - `getOverview(filters?)` — dataset rollup + sample trace ids.
1576
- * - `queryTraces(filters?, limit, offset)` — paginated summaries.
1577
- * - `countTraces(filters?)` — cheap count without materialisation.
1578
- * - `viewTrace(trace_id, perAttrCap)` — full span list, oversized → summary.
1579
- * - `viewSpans(trace_id, span_ids, perAttrCap)` — surgical span fetch.
1580
- * - `searchTrace(trace_id, regex, max_matches)` — bounded regex hits.
1581
- * - `searchSpan(trace_id, span_id, regex, max_matches)` — single-span search.
1582
- *
1583
- * Multiple implementations ship in the core (`OtlpFileTraceStore`).
1584
- * Downstream callers can supply their own — e.g. a DuckDB-backed
1585
- * adapter or an in-memory adapter for tests — by implementing this
1586
- * interface.
1587
- *
1588
- * Filters compose with AND semantics. Empty/undefined fields impose
1589
- * no constraint. `regex_pattern` is the only opt-in raw-bytes scan —
1590
- * implementations may skip it via `count`/`overview` when not set.
1591
- */
1592
-
1593
- interface TraceAnalysisStore {
1594
- getOverview(filters?: TraceAnalystFilters): Promise<DatasetOverview>;
1595
- queryTraces(opts: {
1596
- filters?: TraceAnalystFilters;
1597
- limit: number;
1598
- offset?: number;
1599
- }): Promise<QueryTracesPage>;
1600
- countTraces(filters?: TraceAnalystFilters): Promise<number>;
1601
- viewTrace(opts: {
1602
- trace_id: string;
1603
- /** Override per-attribute byte cap. Defaults to discovery budget. */
1604
- per_attribute_byte_cap?: number;
1605
- }): Promise<ViewTraceResult>;
1606
- viewSpans(opts: {
1607
- trace_id: string;
1608
- span_ids: readonly string[];
1609
- /** Override per-attribute byte cap. Defaults to surgical budget. */
1610
- per_attribute_byte_cap?: number;
1611
- }): Promise<ViewSpansResult>;
1612
- searchTrace(opts: {
1613
- trace_id: string;
1614
- regex_pattern: string;
1615
- /** Hard cap on matches returned. Default 50. */
1616
- max_matches?: number;
1617
- }): Promise<SearchTraceResult>;
1618
- searchSpan(opts: {
1619
- trace_id: string;
1620
- span_id: string;
1621
- regex_pattern: string;
1622
- max_matches?: number;
1623
- }): Promise<SearchSpanResult>;
1624
- }
1625
-
1626
- /**
1627
- * Analyst contract — the missing orchestration layer over agent-eval's
1628
- * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
1629
- * SemanticConceptJudge, JudgeFn, ...).
1630
- *
1631
- * Each existing primitive returns its own output shape. The Analyst
1632
- * contract is the single envelope every primitive lifts into, so a
1633
- * registry can run N analysts against a run and a single renderer can
1634
- * compose findings without knowing which analyzer produced them.
1635
- *
1636
- * The contract is intentionally domain-agnostic: nothing here knows
1637
- * about code, voice, RAG, or any particular agent stack. Analysts
1638
- * declare what INPUT KIND they need (a trace store, an artifact dir,
1639
- * a RunRecord, a JudgeInput, or `custom`), and the registry routes
1640
- * the matching input from `AnalystRunInputs`.
1641
- */
1642
-
1643
- /**
1644
- * Unified envelope every analyst emits. Schema-versioned so renderers
1645
- * and time-series diffs survive future field additions.
1646
- */
1647
- interface AnalystFinding {
1648
- schema_version: '1.0.0';
1649
- /**
1650
- * Stable hash over identity-defining fields (analyst_id + canonical
1651
- * claim + area + optional subject). Two findings from two runs that
1652
- * "are the same finding" share this id — that's what `diffFindings`
1653
- * uses to compute appeared/disappeared sets across runs.
1654
- */
1655
- finding_id: string;
1656
- analyst_id: string;
1657
- produced_at: string;
1658
- severity: AnalystSeverity;
1659
- /**
1660
- * Coarse classification. Renderers group by this. Free-form so
1661
- * domain-specific analysts can introduce categories without a
1662
- * schema change ('agent-reasoning', 'verification', 'cost',
1663
- * 'tool-use', 'safety', 'latency', 'data-quality', ...).
1664
- */
1665
- area: string;
1666
- claim: string;
1667
- rationale?: string;
1668
- evidence_refs: EvidenceRef[];
1669
- recommended_action?: string;
1670
- validation_plan?: string;
1671
- /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */
1672
- confidence: number;
1673
- /**
1674
- * Optional subject the finding is about — leaf id, agent id, request
1675
- * id. Included in finding_id when present so per-subject findings
1676
- * diff cleanly across runs.
1677
- */
1678
- subject?: string;
1679
- /** FIREWALL provenance (docs/learning-flywheel.md): true iff this finding was
1680
- * lifted from a JUDGE verdict (an acceptance score), not OBSERVED from the
1681
- * agent's behavior. A judge-derived finding must NEVER be admitted as a
1682
- * steering input — that is the held-out judge leaking into the loop. Set at
1683
- * the lift site (createJudgeAdapter); checked by `assertNoJudgeVerdict`.
1684
- * Provenance, not evidence presence, is the correct discriminator: an
1685
- * evidence-less trace-analyst observation legitimately steers, while a judge
1686
- * verdict that happens to cite an artifact must not. */
1687
- derived_from_judge?: boolean;
1688
- /** Analyst-private extras; renderers ignore unless they know the analyst. */
1689
- metadata?: Record<string, unknown>;
1690
- }
1691
- type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info';
1692
- interface EvidenceRef {
1693
- /**
1694
- * Where the evidence lives. `span` and `event` refer to OTLP trace
1695
- * elements; `artifact` to a file inside the run's artifact tree;
1696
- * `finding` to another AnalystFinding (cross-analyst chaining);
1697
- * `metric` to a named scalar reading the renderer knows how to read.
1698
- */
1699
- kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
1700
- uri: string;
1701
- excerpt?: string;
1702
- }
1703
- /**
1704
- * The discriminator the registry uses to pass the right input.
1705
- * `custom` is the escape hatch — analysts that need something else
1706
- * (e.g. an embedding cache, a partner SDK handle) read it from
1707
- * `AnalystRunInputs.custom[<analyst id>]`.
1708
- */
1709
- type AnalystInputKind = 'trace-store' | 'artifact-dir' | 'run-record' | 'judge-input' | 'custom';
1710
- interface AnalystCost {
1711
- /** `deterministic` analysts MUST NOT call the LLM. */
1712
- kind: 'deterministic' | 'llm';
1713
- /** Optional declared upper bound; the registry can enforce a budget. */
1714
- est_usd_per_run?: number;
1715
- /** Models the analyst expects to use (informational). */
1716
- models?: string[];
1717
- /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */
1718
- settlement_timeout_ms?: number;
1719
- }
1720
- interface AnalystRequirements {
1721
- /** Min number of shots / samples the analyst needs to produce signal. */
1722
- min_shots?: number;
1723
- /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */
1724
- capabilities?: string[];
1725
- }
1726
- /**
1727
- * What's passed to every analyst call. The registry resolves which
1728
- * field the analyst's `inputKind` selects and asserts it's present.
1729
- */
1730
- interface AnalystRunInputs {
1731
- traceStore?: TraceAnalysisStore;
1732
- artifactDir?: string;
1733
- runRecord?: RunRecord;
1734
- judgeInput?: JudgeInput;
1735
- /** Keyed by analyst id; populated by callers that registered custom analysts. */
1736
- custom?: Record<string, unknown>;
1737
- }
1738
- interface AnalystContext {
1739
- runId: string;
1740
- /** Stable correlation id so logs from a single registry.run() share a tag. */
1741
- correlationId: string;
1742
- /** Enforced wall-clock deadline (epoch ms). */
1743
- deadlineMs?: number;
1744
- /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */
1745
- budgetUsd?: number;
1746
- /** Shared paid-call account when the analyst runs inside a larger campaign. */
1747
- costLedger?: CostLedgerHandle;
1748
- /** Attribution phase used when writing to the shared paid-call account. */
1749
- costPhase?: string;
1750
- /**
1751
- * Shared chat client. Analysts that call an LLM go through this so
1752
- * the operator picks transport (sandbox-sdk | router | cli-bridge |
1753
- * direct-provider | mock) at the registry boundary without touching
1754
- * analyst code.
1755
- */
1756
- chat?: ChatClient;
1757
- /**
1758
- * Findings from a prior run the operator wants the analyst to see as
1759
- * retrieval context. Kinds that take advantage of cross-run memory
1760
- * (failure-mode "I saw this cluster last run", knowledge-gap "the wiki
1761
- * page I asked for is still missing") render these into the actor's
1762
- * working set. Filtering is the operator's job: pass the slice that
1763
- * matches the analyst's id, or pass everything and let the kind
1764
- * filter. Empty / absent means no cross-run context.
1765
- */
1766
- priorFindings?: ReadonlyArray<AnalystFinding>;
1767
- /**
1768
- * Findings emitted by analysts that completed earlier in this registry run.
1769
- * This is separate from `priorFindings`: upstream findings are dependency
1770
- * context for the current pass, while prior findings are cross-run memory.
1771
- * The registry populates this only when `RegistryRunOpts.chainFindings` is on.
1772
- */
1773
- upstreamFindings?: ReadonlyArray<AnalystFinding>;
1774
- /**
1775
- * Report metered work independently of findings. This keeps an empty finding
1776
- * set from erasing token/cost telemetry. Multiple receipts are accumulated.
1777
- */
1778
- recordUsage?: (receipt: AnalystUsageReceipt) => void;
1779
- /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */
1780
- tags?: Record<string, string>;
1781
- /** Logger callback — analysts SHOULD prefer this over console.* for testability. */
1782
- log?: (msg: string, fields?: Record<string, unknown>) => void;
1783
- /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
1784
- signal?: AbortSignal;
1785
- }
1786
- /**
1787
- * The minimal contract. Concrete analysts can refine `TInput` so
1788
- * implementations stay type-safe (e.g. a trace analyst's `TInput` is
1789
- * `TraceAnalysisStore`); the registry passes the right field from
1790
- * `AnalystRunInputs` based on `inputKind`.
1791
- */
1792
- interface Analyst<TInput = unknown> {
1793
- /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */
1794
- readonly id: string;
1795
- /** Human-readable. One sentence. */
1796
- readonly description: string;
1797
- readonly inputKind: AnalystInputKind;
1798
- readonly cost: AnalystCost;
1799
- readonly requires?: AnalystRequirements;
1800
- /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */
1801
- readonly version: string;
1802
- analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>;
1803
- }
1804
- /** Metered work performed by one analyst call. */
1805
- interface AnalystUsageReceipt {
1806
- /** Number of model-usage records observed at the provider boundary. */
1807
- calls: number | null;
1808
- /** Null when the provider did not return token accounting. */
1809
- tokens: RunTokenUsage | null;
1810
- /** Observed, estimated, or explicitly uncaptured dollar cost. */
1811
- cost: RunCostProvenance;
1812
- /** Known lower bound when one or more calls have uncaptured cost. */
1813
- knownCostUsd?: number;
1814
- }
1815
- /**
1816
- * Compute the stable finding_id from the identity-defining fields.
1817
- * Default implementation hashes {analyst_id, area, subject, normalized claim}.
1818
- * Analysts that emit findings whose claim text varies per run (timestamps,
1819
- * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
1820
- * or (b) move the variable part into `rationale`/`metadata` and keep the
1821
- * `claim` static.
1822
- */
1823
- declare function computeFindingId(input: {
1824
- analyst_id: string;
1825
- area: string;
1826
- subject?: string;
1827
- claim: string;
1828
- /** Override the claim for hashing — use when the displayed claim has run-specific bits. */
1829
- id_basis?: string;
1830
- }): string;
1831
- /**
1832
- * Convenience factory: produce a fully-formed AnalystFinding with the
1833
- * id computed automatically. Analyst code stays terse.
1834
- */
1835
- declare function makeFinding(init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {
1836
- id_basis?: string;
1837
- produced_at?: string;
1838
- }): AnalystFinding;
1839
- interface AnalystRunSummary {
1840
- analyst_id: string;
1841
- status: 'ok' | 'skipped' | 'failed';
1842
- /** Why skipped — missing input, budget exceeded, capability unmet. */
1843
- reason?: string;
1844
- findings_count: number;
1845
- latency_ms: number;
1846
- /** Additive model usage and cost provenance for this analyst. */
1847
- usage: AnalystUsageReceipt;
1848
- /** When `status='failed'`: the error class + message, never the full stack. */
1849
- error?: {
1850
- class: string;
1851
- message: string;
1852
- };
1853
- }
1854
- interface AnalystRunResult {
1855
- run_id: string;
1856
- correlation_id: string;
1857
- started_at: string;
1858
- ended_at: string;
1859
- findings: AnalystFinding[];
1860
- per_analyst: AnalystRunSummary[];
1861
- /** Total LLM cost in USD across all analysts in this registry.run(). */
1862
- total_cost_usd: number;
1863
- /**
1864
- * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only
1865
- * the known subtotal and must not be treated as the run's total spend.
1866
- */
1867
- total_cost_provenance?: RunCostProvenance;
1868
- }
1869
- /**
1870
- * Events emitted by `AnalystRegistry.runStream(...)` in real time as
1871
- * the registry executes. UIs subscribe via `for await (const ev of
1872
- * registry.runStream(...))`; `registry.run(...)` is a thin collector
1873
- * over the same stream, so the two surfaces share their invariants.
1874
- *
1875
- * Per-finding events are intentionally omitted — analyzers are batch
1876
- * operations (an Ax actor returns the full `findings:json[]` at the
1877
- * end of the responder), so streaming inside one analyst would only
1878
- * emit partial JSON consumers can't render. The kind-completion event
1879
- * is the right granularity; subscribers wanting per-finding rendering
1880
- * iterate `event.findings` themselves.
1881
- */
1882
- type AnalystRunEvent = {
1883
- type: 'run-started';
1884
- run_id: string;
1885
- correlation_id: string;
1886
- started_at: string;
1887
- /** The ordered list of analyst ids the registry will run. */
1888
- analyst_ids: ReadonlyArray<string>;
1889
- } | {
1890
- type: 'analyst-skipped';
1891
- summary: AnalystRunSummary;
1892
- } | {
1893
- type: 'analyst-started';
1894
- analyst_id: string;
1895
- started_at: string;
1896
- } | {
1897
- type: 'analyst-completed';
1898
- /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */
1899
- summary: AnalystRunSummary;
1900
- findings: ReadonlyArray<AnalystFinding>;
1901
- } | {
1902
- type: 'run-completed';
1903
- result: AnalystRunResult;
1904
- };
1905
-
1906
- /**
1907
- * Adapter factories — lift each existing agent-eval primitive into the
1908
- * Analyst contract without re-implementing it.
1909
- *
1910
- * Five primitives, five factories. Each one:
1911
- * - Builds an Analyst with a stable id (caller chooses; defaults
1912
- * given), a sensible default `inputKind`, a version derived from
1913
- * the wrapped primitive's version + an adapter revision, and an
1914
- * `analyze()` that calls the primitive and lifts its output to
1915
- * AnalystFinding[] using `makeFinding()`.
1916
- * - Maps severities: the existing `Severity` ('critical' | 'major' |
1917
- * 'minor' | 'info') projects onto AnalystSeverity ('critical' |
1918
- * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →
1919
- * 'medium'. Domain analysts that want finer-grained mapping override.
1920
- *
1921
- * Adapters never own state. Calling the same factory twice with the
1922
- * same primitive instance is safe.
1923
- */
1924
-
1
+ import { a as MultiLayerVerifier, l as VerifyOptions, o as Severity } from "../multi-layer-verifier-BHY1gWAc.js";
2
+ import { C as FindingSubjectKind, D as parseFindingSubject, E as findingSubjectGrammarPromptFor, F as createAnalystAi, H as SemanticConceptJudgeInput, O as renderFindingSubject, P as CreateAnalystAiConfig, S as FindingSubject, T as KIND_EXPECTED_SUBJECTS, U as SemanticConceptJudgeOptions, Y as RunTrace, _ as defaultIsMaterial, a as SkillUsageScanConfig, b as FINDING_SUBJECT_KINDS, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as PersistedFinding, h as FindingsStore, i as SkillUsageReport, k as BehavioralMetrics, l as KNOWLEDGE_POISONING_KIND_SPEC, m as FindingsDiff, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as DiffPolicy, q as RunCritic, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as diffFindings, w as FindingSubjectStringSchema, x as FINDING_SUBJECT_SYNTAX, y as FINDING_SUBJECT_GRAMMAR_PROMPT } from "../skill-usage-C_pXm7lP.js";
3
+ import { c as CostLedgerHandle } from "../cost-ledger-Dye6jCgg.js";
4
+ import { o as LlmClientOptions } from "../llm-client-B_nIBlYo.js";
5
+ import { A as ChatClient, B as SandboxSdkTransportOpts, F as CreateChatClientOpts, I as CustomTransportOpts, L as DirectProviderTransportOpts, M as ChatResponse, N as ChatTransport, P as CliBridgeTransportOpts, R as MockTransportOpts, V as createChatClient, j as ChatRequest, k as ChatCallOpts, m as JudgeInput, p as JudgeFn, z as RouterTransportOpts } from "../types-DGsxbAEd.js";
6
+ import { t as TraceAnalysisStore } from "../store-CxJry_cs.js";
7
+ import { A as AnalystRunResult, C as AnalystContext, D as AnalystRequirements, E as AnalystInputKind, F as computeFindingId, I as makeFinding, M as AnalystSeverity, N as AnalystUsageReceipt, O as AnalystRunEvent, P as EvidenceRef, S as Analyst, T as AnalystFinding, _ as RawAnalystEvidenceSchema, a as AnalystRegistryOptions, b as evidenceRefsFromRawFinding, c as CreateTraceAnalystKindOpts, d as createTraceAnalystKind, f as renderPriorFindings, g as RawAnalystEvidence, h as RAW_FINDING_SCHEMA_PROMPT, i as AnalystRegistry, j as AnalystRunSummary, k as AnalystRunInputs, l as TraceAnalystGolden, m as ANALYST_SEVERITIES, n as buildDefaultAnalystRegistry, o as BudgetPolicy, p as renderUpstreamFindings, r as AnalystHooks, s as RegistryRunOpts, t as DefaultAnalystRegistryOptions, u as TraceAnalystKindSpec, v as RawAnalystFinding, w as AnalystCost, x as parseRawFinding, y as RawAnalystFindingSchema } from "../default-registry-CNPo-Vsb.js";
8
+ import { AxFunction } from "@ax-llm/ax";
9
+ //#region src/analyst/adapters.d.ts
1925
10
  declare function liftSeverity(s: Severity): AnalystSeverity;
1926
11
  interface VerifierAdapterOpts<Env> {
1927
- id?: string;
1928
- area?: string;
1929
- verifier: MultiLayerVerifier<Env>;
1930
- /**
1931
- * The verifier expects an `env` per run. Adapters take it from
1932
- * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.
1933
- */
1934
- options?: Omit<VerifyOptions<Env>, 'env'>;
12
+ id?: string;
13
+ area?: string;
14
+ verifier: MultiLayerVerifier<Env>;
15
+ /**
16
+ * The verifier expects an `env` per run. Adapters take it from
17
+ * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.
18
+ */
19
+ options?: Omit<VerifyOptions<Env>, 'env'>;
1935
20
  }
1936
21
  declare function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env>;
1937
22
  interface RunCriticAdapterOpts {
1938
- id?: string;
1939
- area?: string;
1940
- critic?: RunCritic;
1941
- /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */
1942
- threshold?: number;
23
+ id?: string;
24
+ area?: string;
25
+ critic?: RunCritic;
26
+ /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */
27
+ threshold?: number;
1943
28
  }
1944
29
  declare function createRunCriticAdapter(opts?: RunCriticAdapterOpts): Analyst<RunTrace>;
1945
30
  interface JudgeAdapterOpts {
1946
- id?: string;
1947
- area?: string;
1948
- judge: JudgeFn;
1949
- /** Chat client passed to the JudgeFn. */
1950
- chat: ChatClient;
1951
- /** Optional cost classification — most judges call an LLM. */
1952
- cost?: Analyst['cost'];
1953
- /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */
1954
- threshold?: number;
31
+ id?: string;
32
+ area?: string;
33
+ judge: JudgeFn;
34
+ /** Chat client passed to the JudgeFn. */
35
+ chat: ChatClient;
36
+ /** Optional cost classification — most judges call an LLM. */
37
+ cost?: Analyst['cost'];
38
+ /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */
39
+ threshold?: number;
1955
40
  }
1956
41
  declare function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput>;
1957
42
  interface SemanticConceptJudgeAdapterOpts {
1958
- id?: string;
1959
- area?: string;
1960
- /** Registry context owns cancellation and the per-analyst cost ledger. */
1961
- options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>;
1962
- /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */
1963
- settlementTimeoutMs?: number;
43
+ id?: string;
44
+ area?: string;
45
+ /** Registry context owns cancellation and the per-analyst cost ledger. */
46
+ options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>;
47
+ /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */
48
+ settlementTimeoutMs?: number;
1964
49
  }
1965
50
  declare function createSemanticConceptJudgeAdapter(opts?: SemanticConceptJudgeAdapterOpts): Analyst<SemanticConceptJudgeInput>;
1966
-
1967
- interface CreateAnalystAiConfig {
1968
- /** OpenAI-compatible API key forwarded as `Authorization: Bearer`.
1969
- * cli-bridge ignores the value on loopback but Ax requires a non-empty string. */
1970
- apiKey: string;
1971
- /** OpenAI-compatible base URL — e.g. `https://router.tangle.tools/v1` or a
1972
- * cli-bridge loopback. */
1973
- baseUrl?: string;
1974
- /** Additional headers required by the gateway, such as tenant or execution policy. */
1975
- headers?: Record<string, string>;
1976
- /** Model id forwarded to analyst calls. */
1977
- model: string;
1978
- /** Ax provider name. Defaults to the OpenAI-compatible client. */
1979
- provider?: AxAIArgs<unknown>['name'];
1980
- }
1981
- /**
1982
- * Construct the `AxAIService` an analyst kind calls through
1983
- * (`createTraceAnalystKind({ ai })`).
1984
- *
1985
- * Ax's `ai()` pins `config.model` to the OpenAI catalog enum, but every
1986
- * OpenAI-compatible router an analyst points at (router.tangle.tools,
1987
- * cli-bridge) accepts arbitrary model ids (claude-code/sonnet, openai/gpt-5.4,
1988
- * …). Consumers were each re-rolling `ai({ name, apiKey, apiURL, config })`
1989
- * behind an `as (a: any) => any` cast to dodge the enum; this is the one
1990
- * canonical constructor so they don't have to — and don't take a direct
1991
- * `@ax-llm/ax` dependency for it.
1992
- */
1993
- declare function createAnalystAi(config: CreateAnalystAiConfig): AxAIService;
1994
-
1995
- /**
1996
- * Deterministic behavioral metrics over OTLP spans — pure arithmetic, no LLM.
1997
- *
1998
- * These are the model-independent multiplier: the four trace-quality signals a
1999
- * tolerant analyzer (e.g. HALO) re-derives per run inside the model — token
2000
- * growth, output decay, tool monoculture, missing self-verification — computed
2001
- * here once, in TypeScript, with zero model judgment. A finding that falls out
2002
- * of arithmetic is trivially model-agnostic and cannot hallucinate the trend.
2003
- *
2004
- * General, not trace-specific: the detectors key off token trajectories and
2005
- * tool usage present in any agentic OTLP trace, not any one benchmark.
2006
- */
2007
-
2008
- type SuboptimalCode = 'monotonic-input-growth' | 'output-length-decay' | 'single-tool-dependency' | 'no-self-verification';
2009
- interface SuboptimalSignal {
2010
- code: SuboptimalCode;
2011
- severity: 'high' | 'medium' | 'low';
2012
- /** Human-readable claim, with the backing numbers inlined. */
2013
- detail: string;
2014
- /** The exact figures the detector fired on — auditable, no model in the loop. */
2015
- evidence: Record<string, number | string | boolean>;
2016
- }
2017
- interface BehavioralMetrics {
2018
- /** The only trace represented by these metrics; null when spans are empty. */
2019
- traceId: string | null;
2020
- llmCallCount: number;
2021
- /** Causally serial LLM timelines. Parallel branches are never joined. */
2022
- tokenSequences: BehavioralTokenSequence[];
2023
- /** Token values from the longest serial timeline, retained for convenience. */
2024
- inputTokenTrajectory: number[];
2025
- outputTokenTrajectory: number[];
2026
- toolHistogram: Record<string, number>;
2027
- totalToolCalls: number;
2028
- distinctTools: number;
2029
- /** distinct/total tool calls; 1.0 when there are no tool calls. */
2030
- toolDiversityRatio: number;
2031
- hasSelfVerification: boolean;
2032
- signals: SuboptimalSignal[];
2033
- }
2034
- interface BehavioralTokenSequence {
2035
- scopeId: string;
2036
- spanIds: string[];
2037
- inputTokenTrajectory: Array<number | null>;
2038
- outputTokenTrajectory: Array<number | null>;
2039
- }
2040
-
2041
- /**
2042
- * `behavioralAnalyst` — a DETERMINISTIC analyst (cost.kind = 'deterministic',
2043
- * never calls the LLM). It produces the efficiency/behavioral findings a
2044
- * tolerant agentic analyzer (HALO) re-derives per run inside the model —
2045
- * context bloat, output decay, tool monoculture, missing self-verification —
2046
- * directly from arithmetic over spans (`computeTraceMetrics`).
2047
- *
2048
- * Why it matters: these findings are model-agnostic BY CONSTRUCTION (no model
2049
- * in the loop), so they cannot return 0 on a weak model the way the Ax-RLM
2050
- * does — and they are strictly more reliable than HALO, which spends tokens
2051
- * re-deriving the same numbers and can hallucinate the trend. The agentic
2052
- * RLM kinds remain for SEMANTIC findings that genuinely need a model; this
2053
- * analyst owns the behavioral class.
2054
- */
2055
-
51
+ //#endregion
52
+ //#region src/analyst/behavioral-analyst.d.ts
2056
53
  /**
2057
54
  * Map computed signals → structured AnalystFindings. Pure: no LLM, no clock
2058
55
  * dependence beyond `produced_at` (overridable for deterministic tests).
2059
56
  */
2060
57
  declare function deriveEfficiencyFindings(metrics: BehavioralMetrics, opts?: {
2061
- analystId?: string;
2062
- producedAt?: string;
58
+ analystId?: string;
59
+ producedAt?: string;
2063
60
  }): AnalystFinding[];
2064
61
  /** The deterministic behavioral/efficiency analyst (no LLM, any-model). */
2065
62
  declare function behavioralAnalyst(): Analyst<TraceAnalysisStore>;
2066
-
2067
- /**
2068
- * Typed Ax output for analyst findings.
2069
- *
2070
- * Ax binds the field as `findings:json[]` so the provider emits native
2071
- * structured output. At the kind-factory boundary every row is validated
2072
- * before it becomes an `AnalystFinding`.
2073
- *
2074
- * Why not `f.object().array()` directly in the signature? The Ax
2075
- * signature string `question:string -> findings:json[]` already lets
2076
- * the provider emit JSON arrays. A Zod boundary is required either
2077
- * way (the provider can return any JSON), and Zod gives us a single
2078
- * validation surface independent of which Ax version is installed.
2079
- */
2080
-
2081
- declare const ANALYST_SEVERITIES: readonly ["critical", "high", "medium", "low", "info"];
2082
- declare const RawAnalystEvidenceSchema: z.ZodObject<{
2083
- uri: z.ZodString;
2084
- excerpt: z.ZodOptional<z.ZodString>;
2085
- }, z.core.$strict>;
2086
- type RawAnalystEvidence = z.infer<typeof RawAnalystEvidenceSchema>;
2087
- declare const RawAnalystFindingSchema: z.ZodObject<{
2088
- evidence: z.ZodArray<z.ZodObject<{
2089
- uri: z.ZodString;
2090
- excerpt: z.ZodOptional<z.ZodString>;
2091
- }, z.core.$strict>>;
2092
- severity: z.ZodEnum<{
2093
- low: "low";
2094
- high: "high";
2095
- critical: "critical";
2096
- info: "info";
2097
- medium: "medium";
2098
- }>;
2099
- claim: z.ZodString;
2100
- subject: z.ZodOptional<z.ZodString>;
2101
- confidence: z.ZodNumber;
2102
- rationale: z.ZodOptional<z.ZodString>;
2103
- recommended_action: z.ZodOptional<z.ZodString>;
2104
- }, z.core.$strict>;
2105
- type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
2106
- /**
2107
- * Description embedded into the actor prompt so the LLM knows what
2108
- * shape to emit. Kept here so kinds share one source of truth rather
2109
- * than restating the schema in every prompt.
2110
- */
2111
- declare const RAW_FINDING_SCHEMA_PROMPT = "Each finding MUST be a strict JSON object with:\n - severity: \"critical\" | \"high\" | \"medium\" | \"low\" | \"info\"\n - claim: one-sentence statement (max 2000 chars)\n - subject?: one exact subject form listed by this kind; omit rather than guess\n - evidence: REQUIRED non-empty array of {\"uri\": string, \"excerpt\"?: string}. Use real identifiers with span://, event://, artifact://, metric://, or finding://. Include a short exact quote in excerpt when available. If nothing is citable, do not emit the finding.\n - confidence: number 0..1 (0.9+ exact evidence; 0.6-0.8 inferred pattern; <0.5 speculative)\n - rationale?: one or two reasoning sentences\n - recommended_action?: concrete imperative change; omit for descriptive findings\n\nUnknown fields are rejected. Do not emit area; the factory assigns it. Emit [] when there are no findings. Never fabricate evidence.";
2112
- /** Convert raw citations into the public finding evidence envelope. */
2113
- declare function evidenceRefsFromRawFinding(finding: RawAnalystFinding): EvidenceRef[];
2114
- declare function parseRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): RawAnalystFinding | null;
2115
-
2116
- /**
2117
- * Analyst-kind factory — the typed way to define trace analysts.
2118
- *
2119
- * A "kind" is a specialized analyst whose actor prompt, tool subset,
2120
- * and bounded Ax subqueries target one failure-mode lens (failure-mode
2121
- * classification, knowledge gap discovery, knowledge poisoning,
2122
- * self-improvement, ...). Kinds emit findings in the typed
2123
- * `RawAnalystFinding` shape via a JSON-array Ax output; the factory
2124
- * validates each row with Zod and lifts it into `AnalystFinding[]`.
2125
- *
2126
- * Composition rules:
2127
- * - Each kind owns its actor description. No generic "answer this
2128
- * question" prompt — the prompt names the failure lens.
2129
- * - Each kind picks a narrow tool subset from `ANALYST_TOOL_GROUPS`.
2130
- * A kind that never needs full-trace dumps can drop `viewTrace` /
2131
- * `viewSpans` and stay cheap.
2132
- * - Each kind declares its subquery + parallelism budget. Discovery-heavy
2133
- * kinds can fan out more bounded semantic questions than narrow lenses.
2134
- *
2135
- * Optimizer hook: kinds may declare `goldens` — labeled examples used
2136
- * by `AxBootstrapFewShot` / `AxGEPA` to fit the actor
2137
- * description programmatically. Stored on the kind, not the registry,
2138
- * because the right metric is kind-specific.
2139
- */
2140
-
2141
- /**
2142
- * Per-kind specification. The factory turns this into a regular
2143
- * `Analyst<TraceAnalysisStore>` ready for `AnalystRegistry.register()`.
2144
- */
2145
- interface TraceAnalystKindSpec {
2146
- /** Stable id. Appears in finding_id, telemetry, and registry exclusions. */
2147
- id: string;
2148
- /** One-sentence description shown in `registry.list()`. */
2149
- description: string;
2150
- /** Coarse classification stamped on every emitted finding (`failure-mode`, `knowledge-gap`, ...). */
2151
- area: string;
2152
- /** Bump on any breaking change to the actor prompt or output schema. */
2153
- version: string;
2154
- /** Actor system prompt. Must instruct the LLM to emit `findings` per the schema. */
2155
- actorDescription: string;
2156
- /** Tool functions the actor may call. Pick narrow subsets via `ANALYST_TOOL_GROUPS`. */
2157
- buildTools: (store: TraceAnalysisStore) => AxFunction[];
2158
- /** Bounded semantic subqueries. `maxCalls: 0` disables model fan-out. */
2159
- subqueries?: {
2160
- maxCalls: number;
2161
- maxParallel?: number;
2162
- };
2163
- /** Actor turn cap. Default 12. */
2164
- maxTurns?: number;
2165
- /** Runtime char cap. Default 6000. */
2166
- maxRuntimeChars?: number;
2167
- /** Maximum output tokens for every actor and subquery model call. Default 4096. */
2168
- maxOutputTokens?: number;
2169
- /** Cost classification surfaced in `registry.list()` and budget enforcement. */
2170
- cost: AnalystCost;
2171
- /** Per-finding-row hook — kinds may reject / rewrite before lifting. */
2172
- postProcess?: (row: RawAnalystFinding, ctx: AnalystContext) => RawAnalystFinding | null;
2173
- /** Minimum citations per finding. Default 1; rows below it are rejected. */
2174
- minimumEvidenceCitations?: number;
2175
- /** Optional optimizer hook — populated when a kind wants to fit its prompt against labeled examples. */
2176
- goldens?: TraceAnalystGolden[];
2177
- }
2178
- /**
2179
- * One labeled example consumed by Ax optimizers (MIPRO / GEPA / Bootstrap).
2180
- * Each input is the same `{question}` an analyst would receive; `expected`
2181
- * is the ground-truth finding set a fitted prompt should produce on this
2182
- * input. Metric: kind-specific (default: F1 on `finding_id` overlap).
2183
- */
2184
- interface TraceAnalystGolden {
2185
- question: string;
2186
- expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
2187
- }
2188
- interface CreateTraceAnalystKindOpts {
2189
- /** AxAIService bound at registration time. */
2190
- ai: AxAIService;
2191
- /** Required unless `ai` was created by {@link createAnalystAi}. */
2192
- model?: string;
2193
- /** Override the spec's `version` (e.g. when an optimizer has fitted a new prompt). */
2194
- versionSuffix?: string;
2195
- /**
2196
- * Optional two-phase recovery: when the agentic harvest is empty but the
2197
- * actor produced a substantive free-form `report`, extract findings from that
2198
- * prose via a tolerant chat-completions pass (`structureFindings`) — no
2199
- * strict-emission contract, so it works on weak models. Omit to leave the
2200
- * actor's harvest as-is (the report is still surfaced fail-loud either way).
2201
- */
2202
- recovery?: {
2203
- baseUrl: string;
2204
- apiKey?: string;
2205
- model?: string;
2206
- fetchImpl?: typeof fetch;
2207
- };
2208
- /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */
2209
- settlementTimeoutMs?: number;
2210
- }
2211
- /**
2212
- * Build an `Analyst<TraceAnalysisStore>` from a kind spec.
2213
- *
2214
- * Lifts the Ax pipeline once at registration time so the registry
2215
- * gets a stateless analyst. The Ax agent is freshly constructed per
2216
- * `analyze()` call (the agent carries chat-log + usage state we don't
2217
- * want shared across analyst runs).
2218
- */
2219
- declare function createTraceAnalystKind(spec: TraceAnalystKindSpec, opts: CreateTraceAnalystKindOpts): Analyst<TraceAnalysisStore>;
2220
- /**
2221
- * Render a compact prior-findings block the actor reads alongside its
2222
- * brief. Each row is one line so the actor can scan dozens cheaply.
2223
- * The kind's prompt instructs the actor to (a) check whether a new
2224
- * cluster matches a prior `finding_id` (carry the id forward via
2225
- * `id_basis` to keep diffs stable) and (b) raise severity / confidence
2226
- * when a prior finding has reappeared without remediation.
2227
- *
2228
- * Returns the empty string when there are no prior findings — most
2229
- * runs are "first-of-its-kind" and the prompt stays unchanged.
2230
- *
2231
- * Exported for tests + for consumers that build their own actor
2232
- * prompts (e.g. specialized analysts living outside the default kinds).
2233
- */
2234
- declare function renderPriorFindings(prior: AnalystContext['priorFindings']): string;
2235
- /** Render findings produced earlier in this same registry run. */
2236
- declare function renderUpstreamFindings(upstream: AnalystContext['upstreamFindings']): string;
2237
-
2238
- /**
2239
- * AnalystRegistry — orchestrate N analysts against one run.
2240
- *
2241
- * Owns three responsibilities and only three:
2242
- * 1. Registration — ids must be unique; bad registrations fail loudly
2243
- * at register-time, not run-time.
2244
- * 2. Routing — each analyst declares its `inputKind`; the registry
2245
- * picks the matching field from AnalystRunInputs and skips the
2246
- * analyst with a logged reason if it's missing.
2247
- * 3. Isolation — one analyst's exception MUST NOT stop other analysts.
2248
- * Failed analysts produce zero findings + a 'failed' summary row.
2249
- *
2250
- * Cross-cutting concerns (telemetry, error → finding conversion, cost
2251
- * ingestion, storage rotation) live in `AnalystHooks`. Budget shaping
2252
- * (equal split vs weighted vs custom) lives in `BudgetPolicy`. Both
2253
- * have sensible defaults; consumers override only what they need.
2254
- */
2255
-
2256
- interface AnalystHooks {
2257
- /** Before analyze() — last chance to mutate ctx (e.g. inject tags, override budget). */
2258
- onBeforeAnalyze?(args: {
2259
- analyst: Analyst;
2260
- ctx: AnalystContext;
2261
- runId: string;
2262
- }): void | Promise<void>;
2263
- /** After every analyst (ok | failed | skipped). Use for telemetry, ingestion, rotation. */
2264
- onAfterAnalyze?(args: {
2265
- analyst: Analyst;
2266
- summary: AnalystRunSummary;
2267
- findings: AnalystFinding[];
2268
- runId: string;
2269
- }): void | Promise<void>;
2270
- /**
2271
- * On analyst exception. Hook MAY return findings to convert the
2272
- * error into structured findings; the summary still reports 'failed'.
2273
- * Return void to keep the default empty-findings behavior.
2274
- */
2275
- onError?(args: {
2276
- analyst: Analyst;
2277
- error: Error;
2278
- runId: string;
2279
- }): AnalystFinding[] | undefined | Promise<AnalystFinding[] | undefined>;
2280
- /** Once after registry.run() completes. Use for final aggregation, persistence. */
2281
- onComplete?(args: {
2282
- result: AnalystRunResult;
2283
- }): void | Promise<void>;
2284
- }
2285
- interface BudgetPolicy {
2286
- /** Overall USD cap across the registry.run(). */
2287
- totalUsd?: number;
2288
- /** Per-analyst weight for the default allocator. Missing ids get weight 1. */
2289
- weights?: Record<string, number>;
2290
- /**
2291
- * Custom allocator — receives the analyst, remaining/total budget, and
2292
- * the count of analysts that will run. Returns the per-analyst budget
2293
- * (or undefined only when the run has no overall cap). Overrides weights
2294
- * when set.
2295
- */
2296
- allocate?: (args: {
2297
- analyst: Analyst;
2298
- totalUsd: number | undefined;
2299
- remainingUsd: number | undefined;
2300
- runningCount: number;
2301
- }) => number | undefined;
2302
- }
2303
- interface AnalystRegistryOptions {
2304
- /** Shared chat client passed to every LLM analyst via AnalystContext. */
2305
- chat?: ChatClient;
2306
- /** Logger callback. Defaults to a no-op. */
2307
- log?: (msg: string, fields?: Record<string, unknown>) => void;
2308
- /** Hooks invoked around analyze() — observability + customization seam. */
2309
- hooks?: AnalystHooks;
2310
- /** Default budget when run() doesn't override. */
2311
- defaultBudget?: BudgetPolicy;
2312
- }
2313
- interface RegistryRunOpts {
2314
- /** Restrict to a subset of registered analysts by id. */
2315
- only?: string[];
2316
- /** Skip these analysts even if registered. Useful for cheap iteration. */
2317
- skip?: string[];
2318
- /** Budget policy — totalUsd + optional weights/allocator. Falls back to options.defaultBudget. */
2319
- budget?: BudgetPolicy;
2320
- /** Active-work cap for the complete registry run. Model receipt settlement may follow. */
2321
- timeoutMs?: number;
2322
- /** Abort signal — forwarded into every analyst's context. */
2323
- signal?: AbortSignal;
2324
- /** Shared paid-call account forwarded to every analyst. */
2325
- costLedger?: CostLedgerHandle;
2326
- /** Attribution phase for calls written to `costLedger`. */
2327
- costPhase?: string;
2328
- /** Tags echoed into AnalystContext.tags — useful for tracking environment/version in findings. */
2329
- tags?: Record<string, string>;
2330
- /**
2331
- * Prior-run findings made available as retrieval context to every
2332
- * analyst via `ctx.priorFindings`. The registry forwards the slice
2333
- * whose `analyst_id` matches each registered analyst so a kind sees
2334
- * only its own history. Pass `{ '*': findings }` to broadcast to
2335
- * every analyst (useful when several kinds share the same historical
2336
- * context). For findings from this run, use `chainFindings` instead.
2337
- */
2338
- priorFindings?: ReadonlyArray<AnalystFinding> | Record<string, ReadonlyArray<AnalystFinding>>;
2339
- /**
2340
- * Pass findings produced earlier in this registry run to each later analyst
2341
- * via `ctx.upstreamFindings`. Registration order is dependency order.
2342
- * Disabled by default because independent analyst suites must opt in.
2343
- */
2344
- chainFindings?: boolean;
2345
- }
2346
- declare class AnalystRegistry {
2347
- private readonly analysts;
2348
- private readonly options;
2349
- constructor(options?: AnalystRegistryOptions);
2350
- register(analyst: Analyst): void;
2351
- list(): ReadonlyArray<{
2352
- id: string;
2353
- description: string;
2354
- version: string;
2355
- cost: Analyst['cost'];
2356
- }>;
2357
- run(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): Promise<AnalystRunResult>;
2358
- /**
2359
- * Streaming counterpart to `run()`. Emits `AnalystRunEvent` values
2360
- * in real time — `run-started`, then per-analyst `skipped` /
2361
- * `started` / `completed`, then a terminal `run-completed` whose
2362
- * payload is the full `AnalystRunResult`. UIs use this to render
2363
- * progress; persistence consumers use `run()` and read the result.
2364
- *
2365
- * Hooks (`onBeforeAnalyze` / `onAfterAnalyze` / `onError` /
2366
- * `onComplete`) fire as before — streaming is additive, not a hook
2367
- * replacement.
2368
- */
2369
- runStream(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): AsyncGenerator<AnalystRunEvent, void, void>;
2370
- private selectAnalysts;
2371
- private routeInput;
2372
- }
2373
-
2374
- /**
2375
- * `buildDefaultAnalystRegistry` — the canonical analyst suite, so consumers
2376
- * stop hand-wiring `new AnalystRegistry()` + per-kind `createTraceAnalystKind`.
2377
- *
2378
- * The deterministic `behavioralAnalyst` is ALWAYS registered (it needs no
2379
- * model and is model-agnostic by construction). The agentic RLM kinds are
2380
- * registered only when an `ai` service is supplied — so a caller with no LLM
2381
- * still gets the full behavioral/efficiency diagnosis, and the substrate's
2382
- * "any model (including no model)" guarantee holds at the suite level.
2383
- */
2384
-
2385
- interface DefaultAnalystRegistryOptions {
2386
- /** Ax service for the agentic RLM kinds. Omit → only the deterministic analyst. */
2387
- ai?: AxAIService;
2388
- /** Required unless `ai` was created by `createAnalystAi`. */
2389
- model?: string;
2390
- /** Which agentic kinds to register when `ai` is present. Default = the shipped suite. */
2391
- kinds?: readonly TraceAnalystKindSpec[];
2392
- /** Set false to omit the deterministic behavioral analyst (default: include). */
2393
- includeBehavioral?: boolean;
2394
- /** Forwarded to the AnalystRegistry constructor (signal, tags, priorFindings). */
2395
- registry?: AnalystRegistryOptions;
2396
- }
2397
- declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
2398
-
2399
- /**
2400
- * Typed `FindingSubject` — the canonical grammar every analyst kind emits.
2401
- *
2402
- * Background: kind actor prompts have always documented a subject grammar
2403
- * (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`) but the
2404
- * LLM was unconstrained — it could emit `subject: "fix the prompt"`
2405
- * (prose) and downstream adapters routed on `startsWith(...)` would
2406
- * silently skip it. Every per-vertical `ImprovementAdapter` had a
2407
- * routing table that mostly caught nothing.
2408
- *
2409
- * This module fixes that:
2410
- * - `parseFindingSubject(raw)` — returns the typed `FindingSubject`
2411
- * when `raw` matches the grammar, else `null`. Used at the
2412
- * `RawAnalystFindingSchema` boundary so malformed subjects are
2413
- * rejected loudly instead of silently lifted into the registry.
2414
- * - `FindingSubjectKind` — the union of valid locus categories. Each
2415
- * variant carries the typed components downstream adapters resolve
2416
- * against the agent's surface manifest (no string parsing in the
2417
- * adapter).
2418
- * - `FINDING_SUBJECT_GRAMMAR_PROMPT` — single source of truth for the
2419
- * grammar string embedded in kind actor prompts. Drift between
2420
- * prompt and parser is impossible if every kind imports this.
2421
- *
2422
- * The grammar is intentionally NARROW — only loci the substrate's
2423
- * default `ImprovementAdapter` / `KnowledgeAdapter` can act on. A
2424
- * finding with a subject outside this set fails the parser; the kind
2425
- * author either extends the grammar here (and adds adapter routing)
2426
- * or rephrases the prompt to map onto an existing variant.
2427
- *
2428
- * `failure-mode` is the one exception — its subjects are free-form
2429
- * cluster labels, not loci. The schema preserves them as
2430
- * `{ kind: 'cluster', label }` and the adapters skip them (cluster
2431
- * findings are evidence, not actionable mutations).
2432
- */
2433
-
2434
- /**
2435
- * Discriminated union of every locus the substrate can route findings to.
2436
- *
2437
- * Adapters narrow on `kind` and use the typed components (no string
2438
- * parsing). Adding a variant here REQUIRES updating the parser, the
2439
- * grammar prompt, and at least one adapter — by design.
2440
- */
2441
- type FindingSubject = {
2442
- kind: 'knowledge.wiki';
2443
- slug: string;
2444
- heading?: string;
2445
- } | {
2446
- kind: 'knowledge.claim';
2447
- topic: string;
2448
- } | {
2449
- kind: 'knowledge.raw';
2450
- sourceId: string;
2451
- } | {
2452
- kind: 'knowledge.stale';
2453
- slug: string;
2454
- } | {
2455
- kind: 'system-prompt';
2456
- section: string;
2457
- } | {
2458
- kind: 'skill';
2459
- name: string;
2460
- } | {
2461
- kind: 'tool-doc';
2462
- tool: string;
2463
- aspect?: string;
2464
- } | {
2465
- kind: 'new-tool';
2466
- name: string;
2467
- } | {
2468
- kind: 'mcp';
2469
- server: string;
2470
- tool?: string;
2471
- } | {
2472
- kind: 'hook';
2473
- name: string;
2474
- } | {
2475
- kind: 'subagent';
2476
- name: string;
2477
- } | {
2478
- kind: 'workflow';
2479
- name: string;
2480
- } | {
2481
- kind: 'rollout-policy';
2482
- field: string;
2483
- } | {
2484
- kind: 'agent-profile';
2485
- field: string;
2486
- } | {
2487
- kind: 'code';
2488
- path: string;
2489
- } | {
2490
- kind: 'rag';
2491
- corpus: string;
2492
- docId: string;
2493
- } | {
2494
- kind: 'memory';
2495
- key: string;
2496
- } | {
2497
- kind: 'scaffolding';
2498
- concern: string;
2499
- } | {
2500
- kind: 'output-schema';
2501
- field: string;
2502
- } | {
2503
- kind: 'websearch.outdated';
2504
- topic: string;
2505
- } | {
2506
- kind: 'prior-run-summary';
2507
- topic: string;
2508
- } | {
2509
- kind: 'cluster';
2510
- label: string;
2511
- };
2512
- type FindingSubjectKind = FindingSubject['kind'];
2513
- declare const FINDING_SUBJECT_KINDS: ReadonlyArray<FindingSubjectKind>;
2514
- /**
2515
- * Parse a raw subject string emitted by an analyst kind's actor.
2516
- *
2517
- * Returns the typed `FindingSubject` when `raw` matches the grammar,
2518
- * else `null`. Callers use the `null` return as a signal to either
2519
- * (a) reject the finding at parse time (kinds that emit typed loci —
2520
- * knowledge-gap, improvement, knowledge-poisoning) or (b) lift it as
2521
- * a cluster label (failure-mode).
2522
- *
2523
- * Slugs are constrained to `[a-z0-9-]+` (lowercase kebab) to keep file
2524
- * paths sane downstream. Topics / keys / sections allow any non-empty
2525
- * string (free-form for the LLM's voice) but get trimmed.
2526
- *
2527
- * Empty / whitespace-only inputs return `null`. `undefined` returns
2528
- * `null`. Both are surfaced by the caller as a rejected subject.
2529
- */
2530
- declare function parseFindingSubject(raw: string | null | undefined): FindingSubject | null;
2531
- /**
2532
- * Render the parsed subject back to its canonical string form. Inverse
2533
- * of `parseFindingSubject`; useful when the substrate constructs new
2534
- * findings programmatically (e.g. for tests, replays, or
2535
- * `id_basis` carry-forward).
2536
- */
2537
- declare function renderFindingSubject(s: FindingSubject): string;
2538
- /**
2539
- * The grammar text embedded into kind actor prompts. Kinds opt into
2540
- * the subset of variants they emit (e.g. `improvement` excludes the
2541
- * cluster variant; `failure-mode` includes ONLY the cluster variant).
2542
- *
2543
- * Drift between prompt and parser is impossible: every kind imports
2544
- * this constant + the matching `expects` set, and the unit tests below
2545
- * lock the table to the parser.
2546
- */
2547
- declare const FINDING_SUBJECT_SYNTAX: Readonly<Record<FindingSubjectKind, string>>;
2548
- declare const FINDING_SUBJECT_GRAMMAR_PROMPT: string;
2549
- /**
2550
- * The variants each kind is allowed to emit. Used at the kind factory
2551
- * boundary so a knowledge-gap finding can't sneak in a `system-prompt:*`
2552
- * subject (the improvement-analyst's job) and vice versa.
2553
- *
2554
- * `failure-mode` is restricted to `cluster` — the only kind that emits
2555
- * a non-locus subject.
2556
- */
2557
- declare const KIND_EXPECTED_SUBJECTS: Record<string, ReadonlyArray<FindingSubjectKind>>;
2558
- /** Render only the subject forms one analyst kind is permitted to emit. */
2559
- declare function findingSubjectGrammarPromptFor(kindId: string): string;
2560
- /**
2561
- * Zod schema that validates a raw subject string and returns the parsed
2562
- * `FindingSubject`. Embedded in `RawAnalystFindingSchema` via
2563
- * `transform`, so `subject` arrives at the kind factory either as a
2564
- * typed locus or as a parse error attached to a single Zod issue.
2565
- *
2566
- * Optionality is preserved: subjects ARE optional on the wire (some
2567
- * findings are descriptive, not actionable). When present, they MUST
2568
- * parse — emitting a malformed subject is a contract violation, not a
2569
- * soft signal.
2570
- */
2571
- declare const FindingSubjectStringSchema: z.ZodString;
2572
-
2573
- /**
2574
- * FindingsStore — durable persistence for AnalystFinding rows + a diff
2575
- * helper so we can answer "what changed since the last run?" without
2576
- * recomputing analysts.
2577
- *
2578
- * On-disk shape is JSONL: one finding per line, append-only, locked via
2579
- * LockedJsonlAppender. Operators get crash-safety (no partial JSON),
2580
- * cheap reads (sequential parse), and trivial backup (rsync the file).
2581
- *
2582
- * Reads are non-locking: a reader sees a consistent snapshot of all
2583
- * fully-written lines and skips an incomplete trailing line if the
2584
- * writer is mid-append. Cross-process locking is intentionally out of
2585
- * scope (see locked-jsonl-appender.ts).
2586
- *
2587
- * The store is run-scoped: callers pass `runId` on append and on load,
2588
- * which keeps multi-run files cleanly partitioned. The `diffFindings`
2589
- * helper compares two run-id sets using stable `finding_id` semantics —
2590
- * the diff is the cross-run signal the regression dashboard renders.
2591
- */
2592
-
2593
- /**
2594
- * One persisted row. We attach `run_id` on disk so a single file can
2595
- * hold multiple runs and the diff helper can query without re-walking
2596
- * separate files.
2597
- */
2598
- interface PersistedFinding extends AnalystFinding {
2599
- run_id: string;
2600
- }
2601
- declare class FindingsStore {
2602
- readonly path: string;
2603
- private readonly appender;
2604
- constructor(path: string);
2605
- append(runId: string, findings: AnalystFinding[]): Promise<void>;
2606
- /** Load every persisted finding. Discards malformed trailing lines silently. */
2607
- loadAll(): PersistedFinding[];
2608
- /** Filter to a single run. */
2609
- loadRun(runId: string): PersistedFinding[];
2610
- }
2611
- interface FindingsDiff {
2612
- /** New finding ids in `current` that weren't in `previous`. */
2613
- appeared: PersistedFinding[];
2614
- /** Finding ids in `previous` that aren't in `current`. */
2615
- disappeared: PersistedFinding[];
2616
- /** Same finding id present in both runs and unchanged per the materiality test. */
2617
- persisted: PersistedFinding[];
2618
- /**
2619
- * Same finding id in both runs but at least one non-identity field
2620
- * shifted per `DiffPolicy.isMaterial`. Reported as [previous, current].
2621
- */
2622
- changed: Array<{
2623
- previous: PersistedFinding;
2624
- current: PersistedFinding;
2625
- }>;
2626
- }
2627
- interface DiffPolicy {
2628
- /**
2629
- * Predicate that decides whether two findings (same finding_id) count
2630
- * as a material change. Defaults to {@link defaultIsMaterial}: severity
2631
- * shift, confidence Δ > 0.05, or evidence count change. Compliance /
2632
- * perf consumers MAY supply a stricter predicate (e.g. rationale text
2633
- * diff, metric Δ thresholds).
2634
- */
2635
- isMaterial?: (previous: AnalystFinding, current: AnalystFinding) => boolean;
2636
- }
2637
- /**
2638
- * Default materiality test. Deliberately narrow so LLM-reword churn
2639
- * doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.
2640
- */
2641
- declare function defaultIsMaterial(a: AnalystFinding, b: AnalystFinding): boolean;
2642
- /**
2643
- * Diff two findings sets by stable finding_id. Callers typically load
2644
- * the two run-id slices from the same store and pass them in.
2645
- */
2646
- declare function diffFindings(previous: PersistedFinding[], current: PersistedFinding[], policy?: DiffPolicy): FindingsDiff;
2647
-
2648
- /**
2649
- * Failure-mode analyst — classifies what went wrong and why.
2650
- *
2651
- * Brief: read the trace dataset, identify the top failure modes across
2652
- * runs, classify each with severity + evidence, and surface them as
2653
- * findings. The actor's job is *taxonomy + evidence*, not fix-design —
2654
- * that's the improvement-analyst's job.
2655
- *
2656
- * Eight bounded model subqueries let the actor compare candidate
2657
- * clusters in parallel after it has loaded representative evidence.
2658
- */
2659
-
2660
- declare const FAILURE_MODE_KIND_SPEC: TraceAnalystKindSpec;
2661
-
2662
- /**
2663
- * Improvement analyst — actionable self-improvement findings.
2664
- *
2665
- * Brief: read findings from upstream analysts (failure-mode,
2666
- * knowledge-gap, knowledge-poisoning) AND the trace dataset itself,
2667
- * then propose **concrete edits** to the agent's runtime: prompt
2668
- * additions, RAG documents to ingest, tool descriptions to rewrite,
2669
- * scaffolding changes to make, memory entries to invalidate. Each
2670
- * finding is one proposed edit with the locus, the diff, and the
2671
- * expected effect.
2672
- *
2673
- * This is the self-improvement loop's last mile: the prior
2674
- * kinds describe *what's wrong*; this kind describes *what to change*.
2675
- *
2676
- * Eight bounded model subqueries let the actor compare competing fix
2677
- * directions over the same cited evidence before recommending one.
2678
- */
2679
-
2680
- declare const IMPROVEMENT_KIND_SPEC: TraceAnalystKindSpec;
2681
-
2682
- /**
2683
- * Knowledge-gap analyst — what did the agent NOT know that it needed?
2684
- *
2685
- * Brief: find moments in the trace where the agent had to guess, ask
2686
- * the user to fill in context, recover from a wrong assumption, or
2687
- * loop on a retrieval. Each finding names a *missing or outdated piece
2688
- * of knowledge* the agent's curated knowledge base should have held —
2689
- * or a downstream lookup (web, docs, tool description) that surfaced
2690
- * stale or outdated information.
2691
- *
2692
- * The primary expected store is `@tangle-network/agent-knowledge`: a
2693
- * Karpathy-style wiki the agent maintains with raw ↔ curated pages,
2694
- * source anchors, and claim/relation triples. A gap is anything the
2695
- * agent had to discover at run-time that should already have lived
2696
- * there. Secondary loci: web-search results that returned outdated
2697
- * pages, tool descriptions that omitted critical behavior, system-
2698
- * prompt sections that didn't cover the case.
2699
- *
2700
- * Distinct from failure-mode: failure-mode classifies *how* it broke;
2701
- * knowledge-gap names the *information* whose absence (or staleness)
2702
- * caused the break. One failure-mode often maps to several gaps.
2703
- *
2704
- * Five bounded model subqueries let the actor compare candidate gaps
2705
- * across source layers after it has loaded the relevant excerpts.
2706
- */
2707
-
2708
- declare const KNOWLEDGE_GAP_KIND_SPEC: TraceAnalystKindSpec;
2709
-
2710
- /**
2711
- * Knowledge-poisoning analyst — what FALSE information misled the agent?
2712
- *
2713
- * Brief: find moments where the agent acted on information that was
2714
- * *wrong* — stale memory, RAG documents that contradicted ground truth,
2715
- * tool descriptions that lied about return shapes, system-prompt
2716
- * instructions that no longer matched reality, prior-run summaries that
2717
- * cached a wrong decision.
2718
- *
2719
- * Distinct from knowledge-gap: a gap is "the agent didn't know X"; a
2720
- * poisoning is "the agent confidently used X, but X was wrong." Gaps
2721
- * surface as questions / self-correction; poisonings surface as
2722
- * confident-but-wrong actions that downstream evidence contradicts.
2723
- *
2724
- * Eight bounded model subqueries let the actor independently assess
2725
- * the action and contradiction excerpts for candidate poisonings.
2726
- */
2727
-
2728
- declare const KNOWLEDGE_POISONING_KIND_SPEC: TraceAnalystKindSpec;
2729
-
2730
- /**
2731
- * Default analyst kinds focused on agent failure + recursive
2732
- * self-improvement.
2733
- *
2734
- * The four kinds chain: failure-mode classifies; knowledge-gap and
2735
- * knowledge-poisoning explain *why* in two orthogonal ways; improvement
2736
- * proposes concrete edits. Register all four against the same trace
2737
- * store in this order and run the registry with `chainFindings: true`
2738
- * to pass each completed kind's findings to the kinds that follow it.
2739
- */
2740
-
2741
- /**
2742
- * The default kind suite. Order is the run order operators should
2743
- * use: failure-mode first (no upstream deps), gap + poisoning next
2744
- * (both depend on failures), improvement last (chains all three).
2745
- */
2746
- declare const DEFAULT_TRACE_ANALYST_KINDS: readonly TraceAnalystKindSpec[];
2747
-
2748
- /**
2749
- * Skill-usage analyst — a DETERMINISTIC `Analyst` over a Claude/Codex skill
2750
- * library + its trace corpus. Unlike the trace-store kinds (failure-mode,
2751
- * improvement, ...) this kind calls no LLM: it mines real usage and skill
2752
- * structure and emits findings by rule.
2753
- *
2754
- * It exists because the naive "Skill-tool invocation count" lies low — it
2755
- * misses orchestrated sub-dispatch (a leaf skill run BY /pursue or /governor
2756
- * logs under the parent), slash-command entry, local-script bypass, and
2757
- * on-disk artifacts. The 2026-05-30 skill audit found 39/53 skills at zero
2758
- * direct invocations, yet only one was a genuine cut: the rest were
2759
- * measurement-invisible or discovery-limited. This analyst encodes that
2760
- * lesson as a multi-signal usage model so a cheap repeatable pass can keep
2761
- * the library honest, and so the expensive audit workflow's verdicts can
2762
- * GEPA-distill it toward agreement (see `gold/skill-verdicts.gold.jsonl`).
2763
- *
2764
- * Report-building (`buildSkillUsageReport`, an fs scan) is separated from
2765
- * finding emission (`SkillUsageAnalyst.analyze`, pure) so the slow scan runs
2766
- * once at the registry boundary and the rule logic stays unit-testable.
2767
- */
2768
-
2769
- type SkillKind = 'public' | 'private';
2770
- /** One skill's multi-signal usage + structure. All counts are deterministic. */
2771
- interface SkillUsageRecord {
2772
- name: string;
2773
- kind: SkillKind;
2774
- /** Absolute path to the skill's SKILL.md. */
2775
- path: string;
2776
- lines: number;
2777
- /** `"skill":"<name>"` Skill-tool invocations across the trace corpus. */
2778
- directInvocations: number;
2779
- /** `<command-name>/<name>` slash invocations across the trace corpus. */
2780
- slashInvocations: number;
2781
- /** Sibling skills whose SKILL.md dispatches to this one (`/<name>`). Proxy
2782
- * for orchestrated sub-dispatch the per-skill counter cannot see. */
2783
- inboundRefs: number;
2784
- /** On-disk artifacts attributable to the skill (e.g. `.evolve/<name>/**`). */
2785
- artifactCount: number;
2786
- /** Tangle-private reference count in the body (leak signal for public skills). */
2787
- tanglePrivateRefs: number;
2788
- hasReferencesDir: boolean;
2789
- hasEvalsDir: boolean;
2790
- /** Body mentions `skill-runs.jsonl` (visible to /reflect + /governor). */
2791
- logsRuns: boolean;
2792
- /** Description carries an explicit `Triggers:` clause / trigger phrases. */
2793
- hasTriggerPhrases: boolean;
2794
- }
2795
- interface SkillUsageReport {
2796
- generatedFromTraces: number;
2797
- records: SkillUsageRecord[];
2798
- }
2799
- interface SkillUsageScanConfig {
2800
- /** Dirs holding `*.jsonl` transcripts (Claude `~/.claude/projects`, Codex sessions). */
2801
- transcriptDirs: string[];
2802
- /** Skill roots to scan; each dir directly under `root` with a `SKILL.md` is a skill. */
2803
- skillRoots: {
2804
- root: string;
2805
- kind: SkillKind;
2806
- }[];
2807
- /** Roots scanned for `<root>/.evolve/<skill>` artifact dirs. */
2808
- artifactRoots?: string[];
2809
- /** Token-prefixed mappings: skill name → extra artifact subpaths under an artifactRoot
2810
- * (e.g. reflect → `.evolve/reflections`). Catches non-eponymous artifact dirs. */
2811
- artifactAliases?: Record<string, string[]>;
2812
- /** Cap files read per transcript dir (bounds a huge corpus); 0 = unbounded. */
2813
- maxTranscriptsPerDir?: number;
2814
- }
2815
- /** Scan the corpus + skill roots into a {@link SkillUsageReport}. Deterministic. */
2816
- declare function buildSkillUsageReport(config: SkillUsageScanConfig): SkillUsageReport;
2817
- /** Pure rule pass over a report → findings. Exported for direct/unit use. */
2818
- declare function emitSkillUsageFindings(report: SkillUsageReport, producedAt: string): AnalystFinding[];
2819
- declare class SkillUsageAnalyst implements Analyst<SkillUsageReport> {
2820
- readonly id = "skill-usage";
2821
- readonly description = "Deterministic multi-signal skill-usage analysis: flags dead skills, measurement-invisible (orchestrated) usage, discovery gaps, public-repo leaks, bloat, missing evals, and missing run-logging.";
2822
- readonly inputKind: "custom";
2823
- readonly cost: {
2824
- kind: "deterministic";
2825
- est_usd_per_run: number;
2826
- };
2827
- readonly version = "1.0.0";
2828
- analyze(input: SkillUsageReport, ctx: AnalystContext): Promise<AnalystFinding[]>;
2829
- }
2830
- declare const SKILL_USAGE_ANALYST: SkillUsageAnalyst;
2831
-
63
+ //#endregion
64
+ //#region src/analyst/parse-tolerant.d.ts
2832
65
  /**
2833
66
  * Forgiving pre-parse for analyst findings. Weak models routinely emit
2834
67
  * schema-correct content in an unusable wrapper — fenced ```json blocks, a
@@ -2854,7 +87,8 @@ declare function coerceJson(text: string): unknown;
2854
87
  * (`parseRawFinding`) — this only fixes the shape and never invents fields.
2855
88
  */
2856
89
  declare function coerceToFindingRows(raw: unknown): unknown[];
2857
-
90
+ //#endregion
91
+ //#region src/analyst/steer-firewall.d.ts
2858
92
  /** DESCRIPTIVE predicate: does the finding cite at least one observable
2859
93
  * (span/event/artifact) evidence ref. Useful for ranking evidence quality or
2860
94
  * rendering — it is NOT the steer gate. Evidence presence is the WRONG
@@ -2888,77 +122,51 @@ declare function isJudgeVerdict(finding: AnalystFinding): boolean;
2888
122
  * compile-time tripwire on the obvious direct channel.
2889
123
  */
2890
124
  declare function assertNoJudgeVerdict(findings: ReadonlyArray<AnalystFinding>, context?: string): ReadonlyArray<AnalystFinding>;
2891
-
2892
- /**
2893
- * `structureFindings` — the deferred structuring pass (DSPy TwoStepAdapter /
2894
- * HALO `synthesize_traces` analog). The agentic actor reasons FREE-FORM and
2895
- * emits a prose `report` (which any model does reliably); this separate, cheap
2896
- * call's ONLY job is to turn that report into `AnalystFinding[]`. Decoupling
2897
- * reasoning from structuring is what makes the SEMANTIC findings model-agnostic
2898
- * — the reasoning model never has to satisfy a strict typed-array contract
2899
- * while it diagnoses.
2900
- *
2901
- * Forgiving: the response runs through `coerceToFindingRows` (de-fence, lift
2902
- * single→array) before Zod, and on a zero-finding extraction from a substantive
2903
- * report it reasks ONCE with the schema restated. Returns a typed outcome so a
2904
- * legitimate "nothing to report" is distinguishable from a failed extraction
2905
- * (no silent empty).
2906
- */
2907
-
125
+ //#endregion
126
+ //#region src/analyst/structure-findings.d.ts
2908
127
  interface StructureFindingsOptions {
2909
- /** The actor's free-form diagnosis prose. */
2910
- report: string;
2911
- analystId: string;
2912
- /** Coarse classification stamped on every extracted finding. */
2913
- area: string;
2914
- model: string;
2915
- baseUrl: string;
2916
- apiKey?: string;
2917
- /** Optional ledger for direct use. */
2918
- costLedger?: CostLedgerHandle;
2919
- costPhase?: string;
2920
- costTags?: Record<string, string>;
2921
- maxTokens?: number;
2922
- signal?: AbortSignal;
2923
- /** Max reask attempts after a zero/invalid extraction. Default 1. */
2924
- maxReasks?: number;
2925
- /** Apply the caller's normal finding rules before a recovered row is lifted. */
2926
- processRow?: (row: RawAnalystFinding) => RawAnalystFinding | null;
2927
- /** Provenance copied onto every recovered finding. */
2928
- findingMetadata?: Record<string, unknown>;
2929
- /** Test seam: inject a fetch (no network in unit tests). */
2930
- fetchImpl?: LlmClientOptions['fetch'];
128
+ /** The actor's free-form diagnosis prose. */
129
+ report: string;
130
+ analystId: string;
131
+ /** Coarse classification stamped on every extracted finding. */
132
+ area: string;
133
+ model: string;
134
+ baseUrl: string;
135
+ apiKey?: string;
136
+ /** Optional ledger for direct use. */
137
+ costLedger?: CostLedgerHandle;
138
+ costPhase?: string;
139
+ costTags?: Record<string, string>;
140
+ maxTokens?: number;
141
+ signal?: AbortSignal;
142
+ /** Max reask attempts after a zero/invalid extraction. Default 1. */
143
+ maxReasks?: number;
144
+ /** Apply the caller's normal finding rules before a recovered row is lifted. */
145
+ processRow?: (row: RawAnalystFinding) => RawAnalystFinding | null;
146
+ /** Provenance copied onto every recovered finding. */
147
+ findingMetadata?: Record<string, unknown>;
148
+ /** Test seam: inject a fetch (no network in unit tests). */
149
+ fetchImpl?: LlmClientOptions['fetch'];
2931
150
  }
2932
151
  interface StructureFindingsResult {
2933
- findings: AnalystFinding[];
2934
- outcome: 'ok' | 'extraction_failed';
152
+ findings: AnalystFinding[];
153
+ outcome: 'ok' | 'extraction_failed';
2935
154
  }
2936
155
  declare function structureFindings(opts: StructureFindingsOptions): Promise<StructureFindingsResult>;
2937
-
2938
- /**
2939
- * Pre-curated tool subsets for analyst kinds.
2940
- *
2941
- * The full trace-analyst tool set is seven functions. Most kinds only
2942
- * need three or four. Picking from named groups instead of importing
2943
- * the whole bundle keeps every kind's actor-context budget tight and
2944
- * makes "what can this analyst see?" obvious at registration time.
2945
- *
2946
- * Each function in the group keeps its full `name`/`description` from
2947
- * `buildTraceAnalystTools` — we filter, we don't re-implement.
2948
- */
2949
-
156
+ //#endregion
157
+ //#region src/analyst/tool-groups.d.ts
2950
158
  /** Named tool sets. Kinds pass `tools: TRACE_TOOL_GROUPS.failureForensics` etc. */
2951
- type TraceToolGroupName =
159
+ type TraceToolGroupName =
2952
160
  /** All seven tools. Use for open-ended discovery kinds. */
2953
- 'all'
161
+ 'all' |
2954
162
  /** Overview + paginated query + count. No deep reads. Cheap. */
2955
- | 'discovery'
163
+ 'discovery' |
2956
164
  /** Discovery + viewTrace + viewSpans. Deep-read but no regex search. */
2957
- | 'discoveryAndRead'
165
+ 'discoveryAndRead' |
2958
166
  /** Discovery + search tools. For pattern-matching across many traces. */
2959
- | 'discoveryAndSearch'
167
+ 'discoveryAndSearch' |
2960
168
  /** Discovery + viewSpans + searchSpan. Targeted-span work after another kind narrows down. */
2961
- | 'targeted';
169
+ 'targeted';
2962
170
  /**
2963
171
  * Build the tool set for a named group bound to a specific trace store.
2964
172
  *
@@ -2967,5 +175,6 @@ type TraceToolGroupName =
2967
175
  * silently returning all tools would defeat the cost-control point.
2968
176
  */
2969
177
  declare function buildTraceToolsForGroup(group: TraceToolGroupName, store: TraceAnalysisStore): AxFunction[];
2970
-
178
+ //#endregion
2971
179
  export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, type CustomTransportOpts, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type MockTransportOpts, type PersistedFinding, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystGolden, type TraceAnalystKindSpec, type TraceToolGroupName, type VerifierAdapterOpts, assertNoJudgeVerdict, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isJudgeVerdict, isTraceObservable, liftSeverity, makeFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
180
+ //# sourceMappingURL=index.d.ts.map