@tangle-network/agent-eval 0.128.2 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (424) hide show
  1. package/CHANGELOG.md +279 -0
  2. package/README.md +19 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +83 -2932
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -364
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1205
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1710
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -894
  34. package/dist/benchmarks/index.js +2 -59
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6390
  44. package/dist/campaign/index.js +3 -212
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -174
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5605
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1937
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -32
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -617
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3776 -15120
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11185 -11191
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -481
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1298
  196. package/dist/reporting.js +6 -50
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +916 -3596
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2362 -1751
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -1048
  211. package/dist/rollout/index.js +8 -110
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/run-record-BuoE80Dq.js.map +1 -0
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -849
  273. package/dist/supervisor-run/index.js +2 -64
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -251
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1174
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/docs/feature-guide.md +1 -1
  301. package/docs/rollout.md +116 -2
  302. package/package.json +18 -10
  303. package/dist/benchmarks/index.js.map +0 -1
  304. package/dist/campaign/index.js.map +0 -1
  305. package/dist/chunk-2JX3CFMB.js +0 -695
  306. package/dist/chunk-2JX3CFMB.js.map +0 -1
  307. package/dist/chunk-2MKQIFS4.js +0 -183
  308. package/dist/chunk-2MKQIFS4.js.map +0 -1
  309. package/dist/chunk-3RF76KTD.js +0 -84
  310. package/dist/chunk-3RF76KTD.js.map +0 -1
  311. package/dist/chunk-5DTSBUL2.js +0 -159
  312. package/dist/chunk-5DTSBUL2.js.map +0 -1
  313. package/dist/chunk-7ZZMD7UK.js +0 -386
  314. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  315. package/dist/chunk-BOD4O7OF.js +0 -40
  316. package/dist/chunk-BOD4O7OF.js.map +0 -1
  317. package/dist/chunk-BYT7ELPS.js +0 -1553
  318. package/dist/chunk-BYT7ELPS.js.map +0 -1
  319. package/dist/chunk-DJKY2TSY.js +0 -2428
  320. package/dist/chunk-DJKY2TSY.js.map +0 -1
  321. package/dist/chunk-DPUHNQLN.js +0 -232
  322. package/dist/chunk-DPUHNQLN.js.map +0 -1
  323. package/dist/chunk-DRYIUNWY.js +0 -622
  324. package/dist/chunk-DRYIUNWY.js.map +0 -1
  325. package/dist/chunk-EJGRPCO3.js +0 -617
  326. package/dist/chunk-EJGRPCO3.js.map +0 -1
  327. package/dist/chunk-EOSZT7PL.js +0 -2001
  328. package/dist/chunk-EOSZT7PL.js.map +0 -1
  329. package/dist/chunk-EZJEIH2R.js +0 -1559
  330. package/dist/chunk-EZJEIH2R.js.map +0 -1
  331. package/dist/chunk-GGE4NNQT.js +0 -65
  332. package/dist/chunk-GGE4NNQT.js.map +0 -1
  333. package/dist/chunk-HHWE3POT.js +0 -94
  334. package/dist/chunk-HHWE3POT.js.map +0 -1
  335. package/dist/chunk-IHQDPH7D.js +0 -171
  336. package/dist/chunk-IHQDPH7D.js.map +0 -1
  337. package/dist/chunk-JHCHEVET.js +0 -274
  338. package/dist/chunk-JHCHEVET.js.map +0 -1
  339. package/dist/chunk-K4DBDHLK.js +0 -158
  340. package/dist/chunk-K4DBDHLK.js.map +0 -1
  341. package/dist/chunk-K6N6XJJX.js +0 -306
  342. package/dist/chunk-K6N6XJJX.js.map +0 -1
  343. package/dist/chunk-MA6HLL3S.js +0 -65
  344. package/dist/chunk-MA6HLL3S.js.map +0 -1
  345. package/dist/chunk-MAZ26DC7.js +0 -99
  346. package/dist/chunk-MAZ26DC7.js.map +0 -1
  347. package/dist/chunk-MHELPNRP.js +0 -1212
  348. package/dist/chunk-MHELPNRP.js.map +0 -1
  349. package/dist/chunk-NACAGYSY.js +0 -1040
  350. package/dist/chunk-NACAGYSY.js.map +0 -1
  351. package/dist/chunk-NKAGIDE2.js +0 -7633
  352. package/dist/chunk-NKAGIDE2.js.map +0 -1
  353. package/dist/chunk-NPCTHQIO.js +0 -91
  354. package/dist/chunk-NPCTHQIO.js.map +0 -1
  355. package/dist/chunk-NYLOYM6N.js +0 -332
  356. package/dist/chunk-NYLOYM6N.js.map +0 -1
  357. package/dist/chunk-ONWEPEDO.js +0 -57
  358. package/dist/chunk-ONWEPEDO.js.map +0 -1
  359. package/dist/chunk-P5W7RQKK.js +0 -576
  360. package/dist/chunk-P5W7RQKK.js.map +0 -1
  361. package/dist/chunk-P6FYH6K4.js +0 -1161
  362. package/dist/chunk-P6FYH6K4.js.map +0 -1
  363. package/dist/chunk-PBE2LOSS.js +0 -669
  364. package/dist/chunk-PBE2LOSS.js.map +0 -1
  365. package/dist/chunk-PC4UYEBM.js +0 -166
  366. package/dist/chunk-PC4UYEBM.js.map +0 -1
  367. package/dist/chunk-PXE2VKMX.js +0 -140
  368. package/dist/chunk-PXE2VKMX.js.map +0 -1
  369. package/dist/chunk-PZ5AY32C.js +0 -10
  370. package/dist/chunk-PZ5AY32C.js.map +0 -1
  371. package/dist/chunk-RZTMDUO7.js +0 -49
  372. package/dist/chunk-RZTMDUO7.js.map +0 -1
  373. package/dist/chunk-S5YLIBFX.js +0 -136
  374. package/dist/chunk-S5YLIBFX.js.map +0 -1
  375. package/dist/chunk-SZLVEKMJ.js +0 -1446
  376. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  377. package/dist/chunk-T4SQEITX.js +0 -95
  378. package/dist/chunk-T4SQEITX.js.map +0 -1
  379. package/dist/chunk-TBL77AUT.js +0 -355
  380. package/dist/chunk-TBL77AUT.js.map +0 -1
  381. package/dist/chunk-TSN7JT6D.js +0 -1646
  382. package/dist/chunk-TSN7JT6D.js.map +0 -1
  383. package/dist/chunk-TT4KNT67.js +0 -124
  384. package/dist/chunk-TT4KNT67.js.map +0 -1
  385. package/dist/chunk-UB2LOJ6Q.js +0 -4461
  386. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  387. package/dist/chunk-UWZZKKU7.js +0 -237
  388. package/dist/chunk-UWZZKKU7.js.map +0 -1
  389. package/dist/chunk-VBQ3CRKH.js +0 -291
  390. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  391. package/dist/chunk-VGRCHJON.js +0 -163
  392. package/dist/chunk-VGRCHJON.js.map +0 -1
  393. package/dist/chunk-VI2UW6B6.js +0 -162
  394. package/dist/chunk-VI2UW6B6.js.map +0 -1
  395. package/dist/chunk-VLOATJQ2.js +0 -908
  396. package/dist/chunk-VLOATJQ2.js.map +0 -1
  397. package/dist/chunk-VQMK5FMP.js +0 -247
  398. package/dist/chunk-VQMK5FMP.js.map +0 -1
  399. package/dist/chunk-VZSRQ272.js +0 -149
  400. package/dist/chunk-VZSRQ272.js.map +0 -1
  401. package/dist/chunk-WGXIEX7P.js +0 -116
  402. package/dist/chunk-WGXIEX7P.js.map +0 -1
  403. package/dist/chunk-WS3NZZQQ.js +0 -929
  404. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  405. package/dist/chunk-XDWDC2MP.js +0 -695
  406. package/dist/chunk-XDWDC2MP.js.map +0 -1
  407. package/dist/chunk-XPRT64IE.js +0 -766
  408. package/dist/chunk-XPRT64IE.js.map +0 -1
  409. package/dist/chunk-YJBNWCAA.js +0 -1056
  410. package/dist/chunk-YJBNWCAA.js.map +0 -1
  411. package/dist/chunk-ZET2UAYW.js +0 -89
  412. package/dist/chunk-ZET2UAYW.js.map +0 -1
  413. package/dist/chunk-ZUUWPZCV.js +0 -752
  414. package/dist/chunk-ZUUWPZCV.js.map +0 -1
  415. package/dist/control.js.map +0 -1
  416. package/dist/hosted/index.js.map +0 -1
  417. package/dist/matrix/index.js.map +0 -1
  418. package/dist/reporting.js.map +0 -1
  419. package/dist/rollout/index.js.map +0 -1
  420. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  421. package/dist/run-campaign-ISHFZ7FJ.js.map +0 -1
  422. package/dist/supervisor-run/index.js.map +0 -1
  423. package/dist/traces.js.map +0 -1
  424. package/dist/wire/index.js.map +0 -1
@@ -1,2890 +1,67 @@
1
- import { TCloud } from '@tangle-network/tcloud';
2
- import { AxAIArgs, AxAIService, AxFunction } from '@ax-llm/ax';
3
- import { z } from 'zod';
4
-
5
- /**
6
- * Validator-output verdict substrate primitive for "did this output pass,
7
- * and how well?"
8
- *
9
- * Used by:
10
- * - `@tangle-network/agent-eval/matrix` — verdict per cell in the cartesian.
11
- * - `@tangle-network/agent-runtime` — Validator<Output, Verdict = DefaultVerdict>.
12
- * Runtime keeps `Validator` because it's coupled to runtime-shaped
13
- * `ValidationCtx` (iteration, signal, traceEmitter); the verdict TYPE
14
- * itself is a substrate concept and lives here.
15
- *
16
- * Repo layering: agent-eval is the substrate (no upward deps). Both
17
- * agent-runtime and agent-knowledge consume this type FROM agent-eval —
18
- * never the other way around. See CLAUDE.md "Repo layering" for the rule.
19
- */
20
- /**
21
- * Minimal verdict shape — `valid` + `score` are required; `scores` +
22
- * `notes` are optional surface. Validators that need richer shapes
23
- * parameterise `Validator<Output, MyVerdict>` with their own type.
24
- *
25
- * Need structured extras? Extend DefaultVerdict with typed fields — never
26
- * serialize extras into `notes`.
27
- */
28
- interface DefaultVerdict {
29
- /** Whether the output meets the validator's pass criteria. */
30
- valid: boolean;
31
- /** Aggregate score in [0, 1]. Drivers use this for winner selection. */
32
- score: number;
33
- /** Per-dimension scores. Free-form; weighted into `score` by the validator. */
34
- scores?: Record<string, number>;
35
- /** Human-readable rationale; surfaces in trace + final-result `winner.verdict`. */
36
- notes?: string;
37
- }
38
-
39
- /**
40
- * Multi-layer verifier — ordered pipeline of verification layers.
41
- *
42
- * Different contract from {@link JudgeRunner} (which runs parallel
43
- * specs against a sandbox). MultiLayerVerifier is a DAG of layers
44
- * (install → typecheck → build → lint → serve → semantic → …) with
45
- * dependency-based skip, per-layer findings, soft-fail semantics, and
46
- * an aggregated `blendedScore` across all passed layers.
47
- *
48
- * Use when you want:
49
- * - ordered stages where a failing upstream stage skips downstream ones
50
- * - each stage produces rich `findings` (severity + message + evidence)
51
- * - a single composite score across stages with per-stage weights
52
- * - soft-fail stages whose failure doesn't abort the pipeline
53
- *
54
- * Use {@link JudgeRunner} when you want:
55
- * - N independent judges running in parallel against the same artifact
56
- * - no inter-judge dependencies
57
- * - boolean `passed` per judge + overall
58
- *
59
- * Both primitives compose — JudgeRunner can be invoked as a single
60
- * layer inside a MultiLayerVerifier if that suits the caller.
61
- */
62
-
63
- type LayerStatus = 'pass' | 'fail' | 'skipped' | 'error' | 'timeout';
64
- type Severity = 'critical' | 'major' | 'minor' | 'info';
65
- interface Finding {
66
- severity: Severity;
67
- message: string;
68
- evidence?: string;
69
- /** Optional layer name the finding belongs to (set by the verifier if omitted). */
70
- layer?: string;
71
- /**
72
- * Free-form structured payload — used by `multiToolchainLayer` to attach
73
- * `{ adapter: 'pnpm' }`, by judges to attach evidence pointers, etc.
74
- * Renderers MAY interrogate; agent-eval primitives never assume shape.
75
- */
76
- detail?: Record<string, unknown>;
77
- }
78
- interface LayerResult {
79
- layer: string;
80
- status: LayerStatus;
81
- /** Origin of an `error` or `timeout`. Defaults to `execution`. */
82
- errorSource?: 'execution' | 'judge';
83
- /** 0..1 score, optional — layers that don't produce a numeric score omit. */
84
- score?: number;
85
- durationMs: number;
86
- findings: Finding[];
87
- /** Short human-readable summary (one line). */
88
- reason?: string;
89
- /**
90
- * Numeric layer-level diagnostics: error counts, warning counts,
91
- * cyclomatic complexity, total adapter wall-time, etc. Keyed by
92
- * diagnostic name; null = "diagnostic not applicable / not measured."
93
- * Renderers that know the keys can display them; ones that don't,
94
- * ignore. Free-form on purpose — consumers type the value shape in
95
- * their own namespace.
96
- */
97
- diagnostics?: Record<string, number | null>;
98
- /** Any rich per-layer detail — rendered as-is by consumers that know the layer. */
99
- detail?: Record<string, unknown>;
100
- }
101
- interface VerifyContext<Env = unknown> {
102
- /** Per-run opaque context the caller provides. Layers destructure what they need. */
103
- env: Env;
104
- /** Previously-computed results from layers that already ran. */
105
- prior: Record<string, LayerResult>;
106
- /** Signal — if aborted, layers MUST bail within reasonable wall. */
107
- signal: AbortSignal;
108
- }
109
- interface Layer<Env = unknown> {
110
- name: string;
111
- /** Origin assigned when this layer errors or times out. Defaults to `execution`. */
112
- errorSource?: 'execution' | 'judge';
113
- /** Stages that must have `status: 'pass'` before this layer runs. */
114
- dependsOn?: string[];
115
- /**
116
- * Weight in the composite `blendedScore`. Default 1.0. Layers with weight 0
117
- * contribute findings but not score.
118
- */
119
- weight?: number;
120
- /**
121
- * If true, a `fail` status contributes to `blendedScore` (as 0) instead of
122
- * being dropped — use for layers whose failure is a real signal. Default:
123
- * fail drops from numerator + denominator, matching VB's existing semantics.
124
- */
125
- failContributesToScore?: boolean;
126
- /** Optional per-layer wall-cap in ms. Honored by the verifier (AbortSignal). */
127
- capMs?: number;
128
- run: (ctx: VerifyContext<Env>) => Promise<LayerResult> | LayerResult;
129
- }
130
- interface VerifyOptions<Env = unknown> {
131
- env: Env;
132
- /**
133
- * Overall wall cap. Default: sum of layer capMs, or Infinity if any layer
134
- * omits a cap. The verifier short-circuits remaining layers on overall cap.
135
- */
136
- overallCapMs?: number;
137
- /** Called with each layer result as it completes. */
138
- onLayer?: (result: LayerResult) => void;
139
- }
140
- /** Extends the substrate verdict spine: `valid` = `allPass`; `score` is the
141
- * complete task score or 0 when the configured scoring panel was incomplete. */
142
- interface VerificationReport extends DefaultVerdict {
143
- layers: LayerResult[];
144
- passCount: number;
145
- failCount: number;
146
- skippedCount: number;
147
- errorCount: number;
148
- /** True iff the configured scoring panel completed and every layer passed. */
149
- allPass: boolean;
150
- /**
151
- * Diagnostic weighted mean across contributing layers. This may represent a
152
- * partial panel. It is 0 when no layer contributed.
153
- */
154
- blendedScore: number;
155
- /**
156
- * Complete task-quality measurement.
157
- * Present when at least one layer produced a valid score, every other layer
158
- * completed successfully or contributed an explicit scored failure, and no
159
- * result is missing because of a failure, skip, error, or timeout.
160
- * Use this field, not `blendedScore`, when creating task labels.
161
- */
162
- taskScore?: number;
163
- durationMs: number;
164
- startedAt: string;
165
- finishedAt: string;
166
- }
167
- /**
168
- * Ordered DAG of verification layers with dependency-based skipping, per-layer findings, soft-fail semantics, and a blended composite score across all passed layers.
169
- */
170
- declare class MultiLayerVerifier<Env = unknown> {
171
- private readonly layers;
172
- constructor(layers: Layer<Env>[]);
173
- run(opts: VerifyOptions<Env>): Promise<VerificationReport>;
174
- }
175
-
176
- interface RunScore {
177
- success: number;
178
- goalProgress: number;
179
- repoGroundedness: number;
180
- driftPenalty: number;
181
- toolUseQuality: number;
182
- patchQuality: number;
183
- testReality: number;
184
- finalGate: number;
185
- reviewerBlockers: number;
186
- costUsd: number;
187
- wallSeconds: number;
188
- notes?: string[];
189
- }
190
- interface RunScoreWeights {
191
- success: number;
192
- goalProgress: number;
193
- repoGroundedness: number;
194
- driftPenalty: number;
195
- toolUseQuality: number;
196
- patchQuality: number;
197
- testReality: number;
198
- finalGate: number;
199
- reviewerBlockers: number;
200
- costUsd: number;
201
- wallSeconds: number;
202
- }
203
-
204
- type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
205
- type AgentProfileDimensionValue = string | number | boolean | null;
206
- interface AgentProfileSource {
207
- /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
208
- kind: string;
209
- /** sha256 over the canonical source profile object. */
210
- hash: string;
211
- }
212
- interface AgentProfileHarness {
213
- id: string;
214
- version?: string;
215
- hash?: string;
216
- }
217
- interface AgentProfileCell {
218
- schemaVersion: AgentProfileCellSchemaVersion;
219
- cellId: string;
220
- profileId: string;
221
- sourceProfile: AgentProfileSource;
222
- harness?: AgentProfileHarness;
223
- model?: string;
224
- promptHash?: string;
225
- dimensions?: Record<string, AgentProfileDimensionValue>;
226
- }
227
-
228
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
229
- interface BudgetSpec {
230
- tokens?: number;
231
- wallMs?: number;
232
- calls?: number;
233
- usd?: number;
234
- }
235
- interface RunOutcome$1 {
236
- score?: number;
237
- pass?: boolean;
238
- failureClass?: FailureClass;
239
- notes?: string;
240
- }
241
- /**
242
- * Layer — optional classification in a nested build workflow.
243
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
244
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
245
- * `app-runtime`: a run of the generated agent against a domain scenario.
246
- * `meta`: any meta-eval (judge replay, correlation analysis).
247
- */
248
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
249
- interface Run {
250
- runId: string;
251
- /**
252
- * Stable identifier of the scenario being executed.
253
- *
254
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
255
- * input WITHOUT this field, substituting a sensible default
256
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
257
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
258
- * keeps the persisted shape unambiguous for downstream filters + aggregations
259
- * while removing the boilerplate of inventing placeholder ids at the call site.
260
- */
261
- scenarioId: string;
262
- variantId?: string;
263
- datasetVersion?: string;
264
- /** Git SHA of agent code at run time. */
265
- codeSha?: string;
266
- /** Hash of the prompt template + any system prompt. */
267
- promptSha?: string;
268
- /** Model id + date + system-prompt hash, concatenated. */
269
- modelFingerprint?: string;
270
- seed?: number;
271
- /** Arbitrary environment markers (shell, docker version, tz). */
272
- envFingerprint?: Record<string, string>;
273
- /** Version of the redaction rules applied to this run. */
274
- redactionVersion?: string;
275
- /** Parent run in a nested build workflow. A builder run's children are
276
- * app-build runs; those children are app-runtime runs. */
277
- parentRunId?: string;
278
- /** Stable project identifier — groups runs across chats + sessions. */
279
- projectId?: string;
280
- /** Chat/conversation identifier within a project. */
281
- chatId?: string;
282
- /** Layer classification — hint for aggregation; not enforced. */
283
- layer?: RunLayer;
284
- startedAt: number;
285
- endedAt?: number;
286
- status: RunStatus;
287
- outcome?: RunOutcome$1;
288
- budget?: BudgetSpec;
289
- /** Free-form labels for downstream grouping. */
290
- tags?: Record<string, string>;
291
- }
292
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
293
- type SpanStatus = 'ok' | 'error';
294
- interface SpanBase {
295
- spanId: string;
296
- parentSpanId?: string;
297
- runId: string;
298
- kind: SpanKind;
299
- name: string;
300
- startedAt: number;
301
- endedAt?: number;
302
- status?: SpanStatus;
303
- error?: string;
304
- /** Anything not covered by typed fields. Kept deliberately free-form. */
305
- attributes?: Record<string, unknown>;
306
- }
307
- interface Message {
308
- role: 'system' | 'user' | 'assistant' | 'tool';
309
- content: string;
310
- tokens?: number;
311
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
312
- images?: Array<{
313
- artifactId?: string;
314
- url?: string;
315
- mime?: string;
316
- }>;
317
- }
318
- interface LlmSpan extends SpanBase {
319
- kind: 'llm';
320
- model: string;
321
- messages: Message[];
322
- output?: string;
323
- inputTokens?: number;
324
- /** All generated tokens, including the reasoning subset when present. */
325
- outputTokens?: number;
326
- cachedTokens?: number;
327
- cacheWriteTokens?: number;
328
- /** Reasoning-token subset of `outputTokens`. */
329
- reasoningTokens?: number;
330
- costUsd?: number;
331
- finishReason?: string;
332
- }
333
- interface ToolSpan extends SpanBase {
334
- kind: 'tool';
335
- toolName: string;
336
- args: unknown;
337
- /** False when the source observed the call but did not capture its arguments. */
338
- argsCaptured?: boolean;
339
- result?: unknown;
340
- latencyMs?: number;
341
- }
342
- interface RetrievalSpan extends SpanBase {
343
- kind: 'retrieval';
344
- query: string;
345
- hits: Array<{
346
- docId: string;
347
- score: number;
348
- content?: string;
349
- }>;
350
- }
351
- interface JudgeSpan extends SpanBase {
352
- kind: 'judge';
353
- judgeId: string;
354
- /** Span this judgment applies to. */
355
- targetSpanId: string;
356
- dimension: string;
357
- /** Numeric score (free-range; interpretation up to the judge). */
358
- score: number;
359
- rationale?: string;
360
- evidence?: string;
361
- }
362
- interface SandboxSpan extends SpanBase {
363
- kind: 'sandbox';
364
- image?: string;
365
- command?: string;
366
- exitCode?: number;
367
- testsTotal?: number;
368
- testsPassed?: number;
369
- stdoutHash?: string;
370
- stderrHash?: string;
371
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
372
- wallMs?: number;
373
- }
374
- interface GenericSpan extends SpanBase {
375
- kind: 'agent' | 'custom';
376
- }
377
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
378
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
379
- interface TraceEvent {
380
- eventId: string;
381
- runId: string;
382
- spanId?: string;
383
- kind: EventKind;
384
- timestamp: number;
385
- payload: Record<string, unknown>;
386
- }
387
- interface BudgetLedgerEntry {
388
- runId: string;
389
- dimension: keyof BudgetSpec;
390
- limit: number;
391
- consumed: number;
392
- remaining: number;
393
- timestamp: number;
394
- breached: boolean;
395
- /** Span that triggered this entry, if any. */
396
- spanId?: string;
397
- }
398
- interface Artifact {
399
- artifactId: string;
400
- runId: string;
401
- spanId?: string;
402
- contentType: string;
403
- sizeBytes: number;
404
- /** sha256 in hex. */
405
- hash: string;
406
- /** External storage URL (R2, S3, filesystem path). */
407
- storageUrl?: string;
408
- /** Inline content for small blobs — keep under ~64KB. */
409
- inlineContent?: string;
410
- }
411
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
412
-
413
- /**
414
- * Paper-grade RunRecord schema + runtime validator.
415
- *
416
- * Every run that participates in a promotion gate, paper table, or
417
- * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
418
- * fields are exactly those the paper "Two Loops, Three Roles" requires
419
- * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
420
- * holdout split tag. A task score is optional because execution-only records
421
- * must preserve missing labels instead of converting errors into zero quality.
422
- *
423
- * This is intentionally NOT a replacement for the rich `Run` /
424
- * `ProposeReviewReport` / `ScenarioResult` types already in the
425
- * package. Those are runtime structures with full provenance. A
426
- * `RunRecord` is the analysis-time projection — the JSON-friendly
427
- * row you'd put in a parquet file or paste into a notebook.
428
- *
429
- * Validate at the boundary:
430
- *
431
- * const rec = validateRunRecord(rawJson) // throws on missing
432
- * const ok = isRunRecord(rawJson) // boolean check
433
- * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
434
- *
435
- * The validator runs in pure TS — zod is intentionally NOT a
436
- * dependency. Round-trip tested in `tests/run-record.test.ts`.
437
- */
438
-
439
- /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
440
- * combined train+test pool that the optimizer is allowed to read. */
441
- type RunSplitTag = 'search' | 'dev' | 'holdout';
442
- /**
443
- * Explicit execution-lifecycle result for a run.
444
- *
445
- * This is separate from task quality (`outcome`) and failure classification.
446
- * Producers set it only from root-run or process evidence.
447
- */
448
- type RunTerminalOutcome = 'succeeded' | 'failed' | 'cancelled' | 'incomplete' | 'unknown';
449
- interface RunTokenUsage {
450
- input: number;
451
- /** All generated tokens charged as output, including reasoning tokens. */
452
- output: number;
453
- /** Reasoning-token subset of `output`, when the provider reports it. */
454
- reasoning?: number;
455
- /** Prompt tokens served from a provider cache. */
456
- cached?: number;
457
- /** Prompt tokens written into a provider cache. */
458
- cacheWrite?: number;
459
- }
460
- /**
461
- * How a run's USD amount was obtained.
462
- */
463
- type RunCostProvenance = {
464
- kind: 'observed';
465
- usd: number;
466
- } | {
467
- kind: 'estimated';
468
- usd: number;
469
- } | {
470
- kind: 'uncaptured';
471
- usd: null;
472
- };
473
- interface RunJudgeMetadata {
474
- model: string;
475
- promptVersion: string;
476
- /** [0,1] confidence the judge declared. Constant judge confidence
477
- * across many runs is a fallback signal (see `canary.ts`). */
478
- confidence: number;
479
- /** True if the judge degraded to a fallback path (rules-only,
480
- * prior-call cache, etc.). The canary uses this to alert. */
481
- fallback: boolean;
482
- }
483
- /**
484
- * Per-judge / per-dimension breakdown for runs scored by an ensemble of
485
- * judges over a multi-dimensional rubric.
486
- *
487
- * The collapsed `outcome.searchScore` / `holdoutScore` carries the
488
- * composite the gate uses. The full breakdown belongs here so consumers
489
- * can answer "which judge disagreed?", "which dimension dragged the
490
- * composite down?", and "did half the panel fail?" without re-running.
491
- *
492
- * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
493
- * `composite` are convenience projections — derivable but precomputed so
494
- * downstream IRR primitives (`interRaterReliability`,
495
- * `corpusInterRaterAgreement`) and reporters don't pay the same
496
- * aggregation twice.
497
- *
498
- * Fail-loud discipline: judges that errored out land in `failedJudges`
499
- * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
500
- * run); the explicit list makes a partial-failure recorded as such.
501
- */
502
- interface JudgeScoresRecord {
503
- /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
504
- perJudge: Record<string, Record<string, number>>;
505
- /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
506
- perDimMean: Record<string, number>;
507
- /** Composite mean across successful judges. Mirrors the task score only
508
- * when `failedJudges` is empty. */
509
- composite: number;
510
- /** Judges that errored or returned an unparseable verdict. Recorded
511
- * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
512
- * not inferred from missing keys in `perJudge`. */
513
- failedJudges?: string[];
514
- /** Free-form notes the judges emitted (joined across judges or
515
- * first-judge only — consumer's choice). */
516
- notes?: string;
517
- }
518
- interface RunOutcome {
519
- /** Score on the search/optimization split. Optional for holdout-only and
520
- * execution-only records. */
521
- searchScore?: number;
522
- /** Score on the held-out split. Optional for search-only and execution-only
523
- * records. When both scores are absent, the run is explicitly unlabeled. */
524
- holdoutScore?: number;
525
- /** Bag of any other metric the run produced — judge dimensions,
526
- * pass/fail counters, latency stats, etc. Numeric only — keeps
527
- * reporters honest. */
528
- raw: Record<string, number>;
529
- /** Per-judge / per-dim breakdown. Consumers writing ensemble
530
- * judgements populate this; substrate primitives like
531
- * `interRaterReliability` and `corpusInterRaterAgreement` accept
532
- * these records as input. Optional — single-judge or scalar-only
533
- * runs leave it unset. */
534
- judgeScores?: JudgeScoresRecord;
535
- /** Authenticity / realness verdict — did the run build the REAL thing on the
536
- * intended infra, or fake it (see `./authenticity`)? Optional: only domains
537
- * with an authenticity config populate it. Carried in the corpus so the
538
- * flywheel / off-policy learning can optimize for real completion, not gamed
539
- * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
540
- * must not count as a real success regardless of `score`. */
541
- realness?: {
542
- score: number;
543
- gated: boolean;
544
- reason?: string;
545
- };
546
- }
547
- /**
548
- * Mandatory paper-grade fields for a single evaluation run. Optional
549
- * fields are extension points; mandatory fields throw if missing.
550
- *
551
- * Hash discipline:
552
- * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
553
- * model (after any steering bundle merge).
554
- * - `configHash` is the sha256 of the effective run config (model,
555
- * temperature, tools, judges, splits). The pair (promptHash,
556
- * configHash) uniquely identifies an experiment cell.
557
- *
558
- * Model snapshot discipline:
559
- * - `model` MUST encode a snapshot version. Bare aliases like
560
- * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
561
- * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
562
- */
563
- interface RunRecord {
564
- /** UUID for the run. */
565
- runId: string;
566
- /** Logical experiment grouping (a treatment vs a baseline within
567
- * the same sweep should share `experimentId`). */
568
- experimentId: string;
569
- /** Stable identifier for the candidate (variant) being run. The
570
- * promotion gate compares two `candidateId`s on matched items. */
571
- candidateId: string;
572
- /** RNG seed for the run. Always recorded — silent re-seeding is
573
- * the most common cause of non-reproducible numbers. */
574
- seed: number;
575
- /** Model identifier WITH snapshot version. */
576
- model: string;
577
- /** sha256 of the effective prompt (post-steering). */
578
- promptHash: string;
579
- /** sha256 of the effective config. */
580
- configHash: string;
581
- /** Git SHA the harness was run from. */
582
- commitSha: string;
583
- /** End-to-end wall-clock duration in milliseconds. */
584
- wallMs: number;
585
- /** Time spent queued before execution started, if known. */
586
- queueMs?: number;
587
- /** Total USD cost, or null when the producer could not capture one. */
588
- costUsd: number | null;
589
- /** Whether `costUsd` came from billing data, a price calculation, or is unavailable. */
590
- costProvenance: RunCostProvenance;
591
- /** Token usage breakdown. */
592
- tokenUsage: RunTokenUsage;
593
- /** Root-run or process terminal result. Never inferred from a child span. */
594
- terminalOutcome: RunTerminalOutcome;
595
- /** Root-run or process failure reason. Valid only for a failed, cancelled,
596
- * or incomplete terminal result; never populated from a child span. */
597
- terminalFailureReason?: string;
598
- /** Judge-side metadata, if a judge was used. */
599
- judgeMetadata?: RunJudgeMetadata;
600
- /** Per-split scores + raw bag. */
601
- outcome: RunOutcome;
602
- /** Canonical task-failure class drawn from the shared
603
- * `FAILURE_CLASSES` taxonomy. Producers set it only from task-result
604
- * evidence. Execution errors belong in
605
- * `outcome.raw.execution_error_count`. */
606
- failureClass?: FailureClass;
607
- /** Free-form task-failure detail scoped under a non-success
608
- * `failureClass`. It is invalid without that class. */
609
- failureMode?: string;
610
- /** Which split this run was drawn from. */
611
- splitTag: RunSplitTag;
612
- /**
613
- * Stable scenario identifier the run observed or was scored against.
614
- * Comparison primitives match this identity rather than input order.
615
- */
616
- scenarioId: string;
617
- /**
618
- * Canonical identity for the agent profile cell that produced this row:
619
- * profile artifact hash plus optional harness/model/prompt/reporting
620
- * dimensions. Use `agentProfile.cellId` to group persona sweeps and
621
- * longitudinal reports by the complete source profile, not by a loose
622
- * candidate label or opaque config hash.
623
- */
624
- agentProfile?: AgentProfileCell;
625
- }
626
-
627
- /**
628
- * RawProviderSink — first-class persistence for the actual HTTP-level
629
- * request/response bodies of every LLM provider call.
630
- *
631
- * Why this is a separate sink from the structured `LlmSpan`:
632
- *
633
- * - `LlmSpan` records the *intent* — model name, messages, output text,
634
- * usage. It's what dashboards read; it's NOT enough for forensics.
635
- * - When a downstream consumer reports "the verifier used the wrong route"
636
- * or "tokens look right but reasoning was missing," the only way to
637
- * answer is the raw HTTP body. Span fields can lie (a proxy can echo
638
- * a different `model` value than what actually answered); the raw
639
- * response is ground truth.
640
- *
641
- * Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
642
- * matrix runner / BuilderSession sets it up automatically) and every
643
- * request, response, and error is recorded — including retries, with the
644
- * attempt index attached so a flaky call's full event chain is recoverable.
645
- *
646
- * Redaction is enforced at sink time. The default redactor strips
647
- * `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
648
- * payload field whose key matches `apiKey | api_key | bearer | password |
649
- * secret | token` (case-insensitive). Override via the sink constructor or
650
- * the per-call `redactor`. The `redactedFields` array on the persisted
651
- * event lets a reviewer see what was stripped without exposing the values.
652
- */
653
- type RawProviderDirection = 'request' | 'response' | 'error';
654
- interface RawProviderEvent {
655
- /** Stable id. Generated by the sink if omitted. */
656
- eventId: string;
657
- /** Trace context populated by `LlmClient` when the call is wrapped in a span. */
658
- runId?: string;
659
- spanId?: string;
660
- /**
661
- * Logical provider name. Free-form so callers can use whatever id matches
662
- * their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
663
- * omitted, derived from `baseUrl` in `LlmClientOptions`.
664
- */
665
- provider: string;
666
- model: string;
667
- /** Endpoint path, e.g. `'/v1/chat/completions'`. */
668
- endpoint: string;
669
- /** Base URL used for the call (already-normalised — no trailing slash). */
670
- baseUrl: string;
671
- /** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
672
- attemptIndex: number;
673
- direction: RawProviderDirection;
674
- /** Unix ms. */
675
- timestamp: number;
676
- /** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
677
- durationMs?: number;
678
- statusCode?: number;
679
- requestHeaders?: Record<string, string>;
680
- requestBody?: unknown;
681
- responseHeaders?: Record<string, string>;
682
- responseBody?: unknown;
683
- /** Set on `direction: 'error'` events. */
684
- errorMessage?: string;
685
- /** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
686
- redactedFields: string[];
687
- }
688
- interface RawProviderSinkFilter {
689
- runId?: string;
690
- spanId?: string;
691
- direction?: RawProviderDirection;
692
- attemptIndex?: number;
693
- }
694
- interface RawProviderSink {
695
- record(event: RawProviderEvent): Promise<void>;
696
- /** Optional listing — implementations that durably persist (file, db) should support this. */
697
- list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
698
- /** Optional teardown for backed implementations. */
699
- close?(): Promise<void>;
700
- }
701
- type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
702
-
703
- interface RunFilter {
704
- scenarioId?: string;
705
- variantId?: string;
706
- status?: RunStatus;
707
- since?: number;
708
- until?: number;
709
- tag?: {
710
- key: string;
711
- value: string;
712
- };
713
- parentRunId?: string;
714
- projectId?: string;
715
- chatId?: string;
716
- layer?: RunLayer;
717
- }
718
- interface SpanFilter {
719
- runId?: string;
720
- parentSpanId?: string;
721
- kind?: SpanKind;
722
- name?: string;
723
- toolName?: string;
724
- judgeId?: string;
725
- since?: number;
726
- until?: number;
727
- }
728
- interface EventFilter {
729
- runId?: string;
730
- spanId?: string;
731
- kind?: EventKind;
732
- since?: number;
733
- until?: number;
734
- }
735
- interface TraceStore {
736
- appendRun(run: Run): Promise<void>;
737
- updateRun(runId: string, patch: Partial<Run>): Promise<void>;
738
- appendSpan(span: Span): Promise<void>;
739
- updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
740
- appendEvent(event: TraceEvent): Promise<void>;
741
- appendArtifact(artifact: Artifact): Promise<void>;
742
- appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
743
- getRun(runId: string): Promise<Run | undefined>;
744
- listRuns(filter?: RunFilter): Promise<Run[]>;
745
- spans(filter?: SpanFilter): Promise<Span[]>;
746
- events(filter?: EventFilter): Promise<TraceEvent[]>;
747
- budget(runId: string): Promise<BudgetLedgerEntry[]>;
748
- artifacts(runId: string): Promise<Artifact[]>;
749
- }
750
-
751
- interface RunTrace {
752
- run: Run;
753
- spans: Span[];
754
- events: TraceEvent[];
755
- artifacts: Artifact[];
756
- budget: BudgetLedgerEntry[];
757
- }
758
- interface RunCriticOptions {
759
- weights?: Partial<RunScoreWeights>;
760
- driftPatterns?: RegExp[];
761
- }
762
- declare class RunCritic {
763
- private readonly weights?;
764
- private readonly driftPatterns;
765
- constructor(options?: RunCriticOptions);
766
- score(store: TraceStore, runId: string): Promise<RunScore>;
767
- scoreTrace(trace: RunTrace): RunScore;
768
- rank(score: RunScore): number;
769
- private isDrift;
770
- }
771
-
772
- type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
773
- interface CostUsage {
774
- inputTokens: number;
775
- /** Includes reasoning tokens when the provider bills them as output. */
776
- outputTokens: number;
777
- /** Reasoning-token subset of outputTokens, when reported. */
778
- reasoningTokens?: number;
779
- /** Prompt tokens served from a provider cache. */
780
- cachedTokens?: number;
781
- /** Prompt tokens written into a provider cache. */
782
- cacheWriteTokens?: number;
783
- }
784
- interface CostCallBase {
785
- callId: string;
786
- channel: CostChannel;
787
- phase: string;
788
- actor: string;
789
- model: string;
790
- maximumCostUsd?: number;
791
- tags?: Record<string, string>;
792
- timestamp: number;
793
- }
794
- interface PendingCostCall extends CostCallBase {
795
- status: 'pending';
796
- }
797
- interface PendingCostCallView extends PendingCostCall {
798
- state: 'active' | 'late' | 'interrupted';
799
- }
800
- interface CostReceipt extends CostCallBase, CostUsage {
801
- status: 'settled';
802
- costUsd: number;
803
- costUnknown: boolean;
804
- usageUnknown?: boolean;
805
- /** Rates used to estimate cost locally. Absent when cost is provider-reported or unknown. */
806
- pricing?: {
807
- inputUsdPerThousand: number;
808
- cachedInputUsdPerThousand?: number;
809
- cacheWriteUsdPerThousand?: number;
810
- outputUsdPerThousand: number;
811
- };
812
- /** Cost reported by the provider, not a local token-price calculation. */
813
- actualCostUsd?: number;
814
- error?: string;
815
- }
816
- interface CostReceiptInput extends CostUsage {
817
- model: string;
818
- /** Caller-supplied rates for a local estimate when the provider does not report billed cost. */
819
- customTokenPricing?: CustomTokenPricing;
820
- actualCostUsd?: number;
821
- costUnknown?: boolean;
822
- usageUnknown?: boolean;
823
- }
824
- /** Per-million token rates for a model or endpoint not covered by package pricing. */
825
- interface CustomTokenPricing {
826
- /** Non-cached input tokens. */
827
- inputUsdPerMillion: number;
828
- /** Cache-read tokens. Falls back to the normal input rate when omitted. */
829
- cachedInputUsdPerMillion?: number;
830
- /** Cache-creation or cache-write tokens. Falls back to the normal input rate when omitted. */
831
- cacheWriteUsdPerMillion?: number;
832
- outputUsdPerMillion: number;
833
- }
834
- type MaximumCharge = {
835
- externallyEnforcedMaximumUsd: number;
836
- } | ({
837
- customTokenPricing: CustomTokenPricing;
838
- } & Pick<CostUsage, 'inputTokens' | 'outputTokens' | 'cachedTokens' | 'cacheWriteTokens'>) | ({
839
- model: string;
840
- } & CostUsage);
841
- interface RunPaidCallInput<T> {
842
- callId?: string;
843
- channel: CostChannel;
844
- phase: string;
845
- actor: string;
846
- /** Used before a provider receipt exists and on failures without one. */
847
- model?: string;
848
- tags?: Record<string, string>;
849
- signal?: AbortSignal;
850
- /** Provider-enforced dollar maximum, or maximum token usage with known pricing. Required when capped. */
851
- maximumCharge?: MaximumCharge;
852
- /** `callId` can be forwarded as the provider's idempotency key. */
853
- execute(signal: AbortSignal, callId: string): Promise<T>;
854
- receipt(value: T): CostReceiptInput;
855
- receiptFromError?(error: Error): CostReceiptInput | undefined;
856
- }
857
- type PaidCallResult<T> = {
858
- succeeded: true;
859
- callId: string;
860
- value: T;
861
- receipt: CostReceipt;
862
- } | {
863
- succeeded: false;
864
- callId?: string;
865
- error: Error;
866
- receipt?: CostReceipt;
867
- };
868
- interface ChannelRollup {
869
- channel: CostChannel;
870
- calls: number;
871
- inputTokens: number;
872
- outputTokens: number;
873
- reasoningTokens?: number;
874
- cachedTokens: number;
875
- cacheWriteTokens?: number;
876
- costUsd: number;
877
- unpricedCalls: number;
878
- unknownUsageCalls: number;
879
- }
880
- interface CostLedgerSummary {
881
- totalCalls: number;
882
- pendingCalls: number;
883
- unresolvedCalls: number;
884
- reservedCostUsd: number;
885
- inputTokens: number;
886
- outputTokens: number;
887
- reasoningTokens?: number;
888
- cachedTokens: number;
889
- cacheWriteTokens?: number;
890
- totalCostUsd: number;
891
- byChannel: ChannelRollup[];
892
- unpricedModels: string[];
893
- fullyPriced: boolean;
894
- usageComplete: boolean;
895
- accountingComplete: boolean;
896
- incompleteReasons: string[];
897
- }
898
- interface CostLedgerFilter {
899
- channel?: CostChannel;
900
- phase?: string;
901
- tags?: Record<string, string>;
902
- }
903
- interface CostLedgerWaitOptions {
904
- /** Maximum time to wait for active provider calls. Default 5 seconds. */
905
- timeoutMs?: number;
906
- /** Wait only for calls matching this attribution filter. */
907
- filter?: CostLedgerFilter;
908
- }
909
- /** Append-only storage. `append` must atomically reject stale revisions. */
910
- interface CostLedgerPersistence {
911
- read(): {
912
- revision: string;
913
- events: string;
914
- };
915
- append(expectedRevision: string, event: string): string | undefined;
916
- }
917
- interface CostLedgerOptions {
918
- costCeilingUsd?: number;
919
- persistence?: CostLedgerPersistence;
920
- /** Import already-settled receipts without admitting new paid work. */
921
- receipts?: readonly CostReceipt[];
922
- }
923
- /** Run-wide paid-call admission, durable call state, receipts, and summaries. */
924
- declare class CostLedger {
925
- private readonly records;
926
- private readonly activeCallIds;
927
- private readonly lateCallIds;
928
- private readonly idleWaiters;
929
- private completedTasks;
930
- private revision;
931
- private costLimitPersisted;
932
- readonly costCeilingUsd?: number;
933
- private readonly persistence?;
934
- constructor(input?: number | CostLedgerOptions);
935
- runPaidCall<T>(input: RunPaidCallInput<T>): Promise<PaidCallResult<T>>;
936
- /** Wait until every call started by this ledger has produced a durable outcome. */
937
- waitForIdle(options?: CostLedgerWaitOptions): Promise<boolean>;
938
- /** Settle a call left pending by a crashed process after reconciling with the provider. */
939
- reconcile(callId: string, observed: CostReceiptInput, options?: {
940
- error?: string;
941
- }): CostReceipt;
942
- list(filter?: CostLedgerFilter): CostReceipt[];
943
- /** Read pending calls without exposing mutable ledger state. */
944
- listPending(filter?: CostLedgerFilter): PendingCostCallView[];
945
- summary(filter?: CostLedgerFilter): CostLedgerSummary;
946
- markCompleted(count?: number): void;
947
- costPerCompletedTask(): number | null;
948
- private execute;
949
- private captureLateOutcome;
950
- private releaseActiveCall;
951
- private commitOutcome;
952
- private captureFailure;
953
- private commitReceipt;
954
- private resolveMaximum;
955
- private hasIncompleteSettledCall;
956
- private appendRecord;
957
- private ensureCostLimitPersisted;
958
- private appendEvent;
959
- }
960
- /** Public callback surface for a shared cost ledger.
961
- *
962
- * Declaration bundles may expose this type through multiple package subpaths.
963
- * Keeping callback contracts structural lets those subpaths compose while the
964
- * concrete {@link CostLedger} retains its private durable state.
965
- */
966
- type CostLedgerHandle = Pick<CostLedger, Exclude<keyof CostLedger, 'listPending' | 'waitForIdle'>> & Partial<Pick<CostLedger, 'listPending' | 'waitForIdle'>>;
967
-
968
- /**
969
- * LLM client with graceful degrade.
970
- *
971
- * OpenAI-compatible `/v1/chat/completions` client with:
972
- * - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
973
- * - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
974
- * - One retry at temperature 1 when a model explicitly requires it.
975
- * - Graceful json_schema → json_object degrade on 400 with schema-reject body.
976
- * - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
977
- * - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
978
- * directly, cli-bridge subscriptions, and any router that speaks the spec.
979
- *
980
- * Usage:
981
- * const { value, result } = await callLlmJson<MyType>(
982
- * { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
983
- * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
984
- * )
985
- *
986
- * This is THE llm-calling seam for agent-eval primitives that need structured
987
- * output (semantic concept judge, reviewer directives, critic scores). Primitives
988
- * that need free-form text use `callLlm` and parse output themselves.
989
- */
990
-
991
- interface LlmMessage {
992
- role: 'system' | 'user' | 'assistant';
993
- /**
994
- * Either a plain text content string OR a multimodal content array
995
- * (text + image_url parts) for vision-capable models.
996
- */
997
- content: string | Array<{
998
- type: 'text';
999
- text: string;
1000
- } | {
1001
- type: 'image_url';
1002
- image_url: {
1003
- url: string;
1004
- detail?: 'auto' | 'low' | 'high';
1005
- };
1006
- }>;
1007
- }
1008
- type LlmThinkingMode = 'enabled' | 'disabled';
1009
- interface LlmCallRequest {
1010
- model: string;
1011
- messages: LlmMessage[];
1012
- /** Optional JSON-mode response format (response_format: json_object). */
1013
- jsonMode?: boolean;
1014
- /** Optional structured output via JSON Schema. Falls back to json_object on 400. */
1015
- jsonSchema?: {
1016
- name: string;
1017
- schema: Record<string, unknown>;
1018
- };
1019
- temperature?: number;
1020
- maxTokens?: number;
1021
- /** OpenAI-compatible reasoning mode. Omitted when the provider default should apply. */
1022
- thinking?: LlmThinkingMode;
1023
- /** Per-call timeout, default 300s. */
1024
- timeoutMs?: number;
1025
- }
1026
- interface LlmUsage {
1027
- promptTokens: number;
1028
- completionTokens: number;
1029
- totalTokens: number;
1030
- /** False when the provider omitted or malformed prompt/completion usage. */
1031
- captured?: boolean;
1032
- /** Reasoning-token subset of completionTokens, when reported. */
1033
- reasoningTokens?: number;
1034
- /** Proxies populate this when prompt caching is on. */
1035
- cachedPromptTokens?: number;
1036
- }
1037
- interface LlmCallResult {
1038
- /** The text content of the first choice. Empty string if none. */
1039
- content: string;
1040
- usage: LlmUsage;
1041
- /**
1042
- * Cost in USD. Uses the provider's reported cost when present, otherwise
1043
- * caller-supplied token pricing. `null` when neither is available.
1044
- */
1045
- costUsd: number | null;
1046
- /** Model name actually used (echoed from response). */
1047
- model: string;
1048
- /** Wall-clock duration of the HTTP call (last attempt, if retried). */
1049
- durationMs: number;
1050
- /**
1051
- * `finish_reason` echoed from the first choice (`stop`, `length`,
1052
- * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
1053
- * Exposed so a free-form `callLlm` caller CAN detect a truncated answer
1054
- * (`length`) instead of treating a cut-off completion as complete. Note:
1055
- * `callLlm` does not itself reject on it — acting on this signal is the
1056
- * caller's responsibility (in-repo free-form drivers do not yet enforce it).
1057
- */
1058
- finishReason?: string | null;
1059
- /**
1060
- * True when `content.trim()` is empty. An empty completion is a silent zero
1061
- * for free-form `callLlm` callers; this flag is the signal a caller can
1062
- * inspect to fail loud rather than proceed on an empty string. `callLlm`
1063
- * surfaces it but does not throw on it.
1064
- */
1065
- contentEmpty?: boolean;
1066
- /** Raw response body. */
1067
- raw: Record<string, unknown>;
1068
- }
1069
- interface LlmClientOptions {
1070
- /** Base URL (without trailing slash). Must end at the `/v1` prefix. */
1071
- baseUrl?: string;
1072
- /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
1073
- apiKey?: string;
1074
- bearer?: string;
1075
- /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
1076
- authHeader?: {
1077
- name: string;
1078
- value: string;
1079
- };
1080
- /** Stable provider idempotency key, reused across retries of this logical call. */
1081
- idempotencyKey?: string;
1082
- /** Default timeout in ms. Per-call can override. */
1083
- defaultTimeoutMs?: number;
1084
- /**
1085
- * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
1086
- * each attempt's per-attempt timeout controller, so aborting it cancels
1087
- * the in-flight fetch. A caller abort is FATAL: it is not retried even
1088
- * though an AbortError otherwise matches the transient patterns.
1089
- */
1090
- signal?: AbortSignal;
1091
- /**
1092
- * Cross-attempt wall-clock budget in ms, measured from the first attempt.
1093
- * Before launching each attempt the loop checks the remaining budget and
1094
- * stops retrying once it is exhausted, rather than waiting the full
1095
- * per-attempt timeout on every retry. Bounds total time independent of
1096
- * total attempts × `timeoutMs`.
1097
- */
1098
- deadlineMs?: number;
1099
- /** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
1100
- maxRetries?: number;
1101
- /** Token rates used when the provider omits cost or package pricing does not cover the model. */
1102
- customTokenPricing?: CustomTokenPricing;
1103
- /**
1104
- * Transport for requests that declare `jsonSchema`. `native` sends
1105
- * `response_format: json_schema`; `json-object` sends the broadly supported
1106
- * JSON mode and relies on the caller to include the schema in model-visible
1107
- * instructions. Default: `native`.
1108
- */
1109
- jsonSchemaTransport?: 'native' | 'json-object';
1110
- /**
1111
- * JSON payload parsing policy. `extract` accepts fenced or prose-prefixed JSON.
1112
- * `exact` requires the complete response content to be one JSON value.
1113
- * Default: `extract`.
1114
- */
1115
- jsonPayloadMode?: 'extract' | 'exact';
1116
- /** Default provider reasoning mode. A per-call request value takes precedence. */
1117
- thinking?: LlmThinkingMode;
1118
- /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
1119
- fetch?: typeof fetch;
1120
- /**
1121
- * Optional raw HTTP capture sink. When provided, every request, response,
1122
- * and error (across all retry attempts) is recorded to the sink, with auth
1123
- * headers and credential-shaped body fields redacted by default. This is
1124
- * the layer-1 forensics primitive: structured `LlmSpan`s record intent,
1125
- * raw events record what actually crossed the wire.
1126
- */
1127
- rawSink?: RawProviderSink;
1128
- /**
1129
- * Logical provider id attached to raw events. When omitted, derived from
1130
- * `baseUrl` via `providerFromBaseUrl`.
1131
- */
1132
- provider?: string;
1133
- /** Trace context attached to raw events; populated by emitter-aware callers. */
1134
- traceContext?: {
1135
- runId?: string;
1136
- spanId?: string;
1137
- };
1138
- /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
1139
- redactor?: ProviderRedactor;
1140
- }
1141
-
1142
- /**
1143
- * Semantic concept judge — "does the built artifact actually implement
1144
- * the features the user asked for?"
1145
- *
1146
- * Distinct from the domain/code/coherence judges in `judges.ts`:
1147
- * - those judges score free-form conversational agent outputs along
1148
- * quality dimensions (accuracy, depth, etc.)
1149
- * - this judge scores a *built artifact* (served HTML + source files)
1150
- * against an explicit list of expected concepts, returning per-concept
1151
- * {present, score 0-10, evidence, severity}.
1152
- *
1153
- * The judge is strict about distinguishing (a) a working implementation
1154
- * from (b) a keyword-present stub. "// TODO: mint button" is NOT present.
1155
- * Only real, functional, wired-up code counts.
1156
- *
1157
- * Use via {@link createSemanticConceptJudge} or directly via
1158
- * {@link runSemanticConceptJudge}. Soft-fails (available=false) on LLM
1159
- * or JSON-parse errors so the caller can treat that as "layer skipped"
1160
- * rather than "layer failed" in a multi-layer pipeline.
1161
- */
1162
-
1163
- /**
1164
- * Implementation complexity class for weighted scoring.
1165
- *
1166
- * - `render` (default): the concept is a UI surface that displays static
1167
- * data — render a list, show a counter, lay out a button. Single-file
1168
- * work, no external integration.
1169
- * - `integrate`: the concept requires wiring a real external system —
1170
- * wallet connect (wagmi + RainbowKit + chain config), payment provider
1171
- * (Stripe Elements + intent + webhook), an API client with auth.
1172
- * Multi-file, library-knowledge, runtime correctness matters.
1173
- * - `compute`: the concept requires algorithmic work — solver, simulator,
1174
- * constraint propagation, ML inference. Correctness > UI polish.
1175
- *
1176
- * Default weights (when applied via `weightConcepts: 'complexity'`):
1177
- * render=1.0, integrate=2.0, compute=2.5
1178
- *
1179
- * Cross-vertical scoring without complexity weighting silently inflates
1180
- * the rate of UI-heavy verticals (healthcare, fintech dashboards) vs
1181
- * integration-heavy verticals (DeFi, wallets) — all concepts treated
1182
- * equally even though the agent does 2-3x the work for `integrate`.
1183
- */
1184
- type ConceptComplexity = 'render' | 'integrate' | 'compute';
1185
- interface ConceptSpec {
1186
- name: string;
1187
- /** Short hints that help the judge; not used for matching. */
1188
- keywords?: string[];
1189
- /** Optional explicit weight; default 1.0. Overrides complexity-derived weight. */
1190
- weight?: number;
1191
- /** Implementation complexity class. Default `render`. */
1192
- complexity?: ConceptComplexity;
1193
- }
1194
- interface SemanticConceptJudgeInput {
1195
- /** Full natural-language prompt the agent was handed. */
1196
- userRequest: string;
1197
- /** Rendered HTML the preview returns (UI artifacts). Optional. */
1198
- servedHtml?: string;
1199
- /** Top-level source files from the agent's workdir. */
1200
- sourceFiles: Array<{
1201
- path: string;
1202
- content: string;
1203
- }>;
1204
- /** The expected concept list. */
1205
- expectedConcepts: ConceptSpec[];
1206
- /** Free-form metadata (id, difficulty) to inject into the prompt. */
1207
- artifactLabel?: string;
1208
- artifactDescription?: string;
1209
- }
1210
- /**
1211
- * Score-aggregation strategy. `mean` averages 0-10 scores uniformly.
1212
- * `complexity` applies the default weight table (render=1, integrate=2,
1213
- * compute=2.5) unless a concept has an explicit `weight`. `explicit`
1214
- * honors only `weight` (defaulting to 1 for unspecified).
1215
- */
1216
- type ConceptWeightStrategy = 'mean' | 'complexity' | 'explicit';
1217
- interface SemanticConceptJudgeOptions {
1218
- /** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */
1219
- model?: string;
1220
- /** Per-call timeout. Default 300s. */
1221
- timeoutMs?: number;
1222
- /** Provider-enforced output limit. Default 16000. */
1223
- maxTokens?: number;
1224
- /** Pipeline budget for the prompt (source blob truncation). Default 45000. */
1225
- maxSourceChars?: number;
1226
- /** Per-file cap before inclusion. Default 20000. */
1227
- maxPerFileChars?: number;
1228
- /** HTML cap. Default 30000. */
1229
- maxHtmlChars?: number;
1230
- /** LlmClient config (baseUrl, apiKey, authHeader, …). */
1231
- llm?: LlmClientOptions;
1232
- costLedger?: CostLedgerHandle;
1233
- costPhase?: string;
1234
- costTags?: Record<string, string>;
1235
- signal?: AbortSignal;
1236
- /**
1237
- * Score aggregation strategy. Default `mean` — uniform average across
1238
- * concepts. Cross-vertical comparisons should use `complexity` to
1239
- * neutralize the integrate-vs-render asymmetry.
1240
- */
1241
- weightConcepts?: ConceptWeightStrategy;
1242
- /** Override the default complexity → weight table. */
1243
- complexityWeights?: Partial<Record<ConceptComplexity, number>>;
1244
- }
1245
-
1246
- interface Scenario {
1247
- id: string;
1248
- persona: string;
1249
- label: string;
1250
- thesis: string;
1251
- dimensions: string[];
1252
- turns: Turn[];
1253
- artifactChecks: ArtifactCheck[];
1254
- systemPromptAppend?: string;
1255
- }
1256
- interface Turn {
1257
- user: string;
1258
- expectedBehaviors: string[];
1259
- adversarial?: boolean;
1260
- feedbackType?: 'correction' | 'rejection' | 'vague' | 'contradictory' | 'escalation';
1261
- }
1262
- interface ArtifactCheck {
1263
- type: 'vault_file_exists' | 'vault_file_contains' | 'block_extracted' | 'code_valid' | 'generation_produced' | 'tool_created' | string;
1264
- target: string;
1265
- contains?: string;
1266
- minCount?: number;
1267
- description: string;
1268
- }
1269
- interface TurnResult {
1270
- turnIndex: number;
1271
- userMessage: string;
1272
- agentResponse: string;
1273
- durationMs: number;
1274
- blocksExtracted: {
1275
- type: string;
1276
- title: string;
1277
- }[];
1278
- containsCode: boolean;
1279
- containsToolCall: boolean;
1280
- }
1281
- interface JudgeScore {
1282
- judgeName: string;
1283
- dimension: string;
1284
- score: number;
1285
- reasoning: string;
1286
- evidence?: string;
1287
- }
1288
- interface CollectedArtifacts {
1289
- vaultFiles: {
1290
- path: string;
1291
- content: string;
1292
- }[];
1293
- blocksExtracted: {
1294
- type: string;
1295
- fields: Record<string, string>;
1296
- }[];
1297
- codeBlocks: {
1298
- language: string;
1299
- code: string;
1300
- }[];
1301
- toolCalls: string[];
1302
- }
1303
- interface JudgeInput {
1304
- scenario: Scenario;
1305
- turns: TurnResult[];
1306
- artifacts: CollectedArtifacts;
1307
- /** Shared ledger for paid built-in judges. Direct calls default to an uncapped ledger. */
1308
- costLedger?: CostLedgerHandle;
1309
- costPhase?: string;
1310
- costTags?: Record<string, string>;
1311
- signal?: AbortSignal;
1312
- /** Exact maximum provider attempts configured on the supplied TCloud client. */
1313
- tcloudMaximumAttempts?: number;
1314
- }
1315
- type JudgeFn = (tc: TCloud, input: JudgeInput) => Promise<JudgeScore[]>;
1316
-
1317
- /**
1318
- * Shared types for the trace-analyst module.
1319
- *
1320
- * Wire format. The store interface speaks `OtlpSpanLike` rows — one JSONL
1321
- * line per span, OTLP-shaped. We do NOT depend on a specific tracing
1322
- * vendor at the type level. Adapter
1323
- * layers map upstream shapes onto this interface.
1324
- *
1325
- * Design constraint. Every read operation that can return arbitrary
1326
- * payload must carry a byte budget so the agent's tool result stays
1327
- * bounded regardless of input trace size. Oversized responses
1328
- * substitute a deterministic summary instead of bytes — see
1329
- * `ViewTraceOversized`.
1330
- */
1331
- /** OTLP span kind (subset we actually use). */
1332
- type TraceAnalystSpanKind = 'AGENT' | 'LLM' | 'TOOL' | 'CHAIN' | 'EVALUATOR' | 'GUARDRAIL' | 'SPAN' | 'UNKNOWN';
1333
- type TraceAnalystSpanStatus = 'OK' | 'ERROR' | 'UNSET';
1334
- /** Subset of OTLP span fields the analyst exposes to the agent. The
1335
- * store's job is to project upstream's full span shape down to this
1336
- * view — the analyst never sees vendor extensions directly. */
1337
- interface TraceAnalystSpan {
1338
- trace_id: string;
1339
- span_id: string;
1340
- parent_span_id: string | null;
1341
- name: string;
1342
- kind: TraceAnalystSpanKind;
1343
- start_time: string;
1344
- end_time: string;
1345
- duration_ms: number;
1346
- status: TraceAnalystSpanStatus;
1347
- status_message?: string;
1348
- service_name: string | null;
1349
- agent_name: string | null;
1350
- model_name: string | null;
1351
- tool_name: string | null;
1352
- /** Raw JSON-serialisable attribute map. May contain large strings;
1353
- * callers must respect the per-attribute byte cap. */
1354
- attributes: Record<string, unknown>;
1355
- }
1356
- interface TraceAnalystTraceSummary {
1357
- trace_id: string;
1358
- service_name: string | null;
1359
- agent_name: string | null;
1360
- span_count: number;
1361
- has_errors: boolean;
1362
- start_time: string;
1363
- end_time: string;
1364
- duration_ms: number;
1365
- raw_jsonl_bytes: number;
1366
- models: string[];
1367
- tools: string[];
1368
- }
1369
- interface TraceAnalystFilters {
1370
- /** Restrict to traces that contain at least one error span. */
1371
- has_errors?: boolean;
1372
- /** Match if any span's `service.name` is in this list. */
1373
- service_names?: string[];
1374
- /** Match if any span's `agent.name` is in this list. */
1375
- agent_names?: string[];
1376
- /** Match if any LLM span's `llm.model_name` is in this list. */
1377
- model_names?: string[];
1378
- /** Match if any tool span's `tool.name` is in this list. */
1379
- tool_names?: string[];
1380
- /** ISO-8601 lower bound on the trace's earliest start time. */
1381
- start_time_after?: string;
1382
- /** ISO-8601 upper bound on the trace's earliest start time. */
1383
- start_time_before?: string;
1384
- /** Single regex applied to raw JSONL bytes for the trace. Opt-in;
1385
- * expensive on large datasets. Use the indexed filters above first. */
1386
- regex_pattern?: string;
1387
- }
1388
- /** One distinct error signature across the dataset — the deterministic unit of
1389
- * failure coverage. Signatures normalize volatile tokens (digits, hex/uuids,
1390
- * paths, durations) out of the span `status_message` so semantically identical
1391
- * failures collapse into one cluster. An analyst that accounts for every
1392
- * cluster has, by construction, covered every distinct failure mode. */
1393
- interface ErrorCluster {
1394
- /** Normalized status_message — the cluster key. */
1395
- signature: string;
1396
- /** A verbatim, un-normalized exemplar message (for exact-string citation). */
1397
- status_message_sample: string;
1398
- /** The span name that most often carries this signature, if any. */
1399
- span_name: string | null;
1400
- /** The tool that most often carries this signature, if any. */
1401
- tool_name: string | null;
1402
- trace_count: number;
1403
- span_count: number;
1404
- /** trace_count / total error traces in the matched set (0..1). */
1405
- prevalence: number;
1406
- /** Real trace ids carrying this signature (capped), passable to view/search. */
1407
- exemplar_trace_ids: string[];
1408
- /** Real span ids carrying this signature (capped). */
1409
- exemplar_span_ids: string[];
1410
- }
1411
- interface DatasetOverview {
1412
- total_traces: number;
1413
- raw_jsonl_bytes: number;
1414
- services: string[];
1415
- agents: string[];
1416
- models: string[];
1417
- tool_names: string[];
1418
- /** Up to 20 real trace ids the agent may pass to view/search tools. */
1419
- sample_trace_ids: string[];
1420
- errors: {
1421
- trace_count: number;
1422
- span_count: number;
1423
- };
1424
- /** The COMPLETE deterministic error-signature population, sorted by
1425
- * trace_count desc. This is the failure-coverage checklist: an analysis is
1426
- * complete only when every cluster here is accounted for. Empty when the
1427
- * matched set has no error spans. */
1428
- error_clusters: ErrorCluster[];
1429
- time_range: {
1430
- earliest: string;
1431
- latest: string;
1432
- } | null;
1433
- }
1434
- interface QueryTracesPage {
1435
- traces: TraceAnalystTraceSummary[];
1436
- total: number;
1437
- has_more: boolean;
1438
- }
1439
- /** Full-trace view. When the response would exceed the per-call byte
1440
- * budget, `oversized` is populated INSTEAD of `spans` so the agent
1441
- * knows to switch to `searchTrace` / `viewSpans`. */
1442
- interface ViewTraceResult {
1443
- trace_id: string;
1444
- spans?: TraceAnalystSpan[];
1445
- oversized?: ViewTraceOversized;
1446
- }
1447
- interface ViewTraceOversized {
1448
- span_count: number;
1449
- /** Names with their counts, sorted desc. Capped at 20 entries. */
1450
- top_span_names: Array<[string, number]>;
1451
- /** Largest single span body (bytes after attribute-cap projection). */
1452
- span_response_bytes_max: number;
1453
- error_span_count: number;
1454
- }
1455
- interface ViewSpansResult {
1456
- trace_id: string;
1457
- spans: TraceAnalystSpan[];
1458
- /** Number of requested span ids that were not found in the trace. */
1459
- missing_span_ids: string[];
1460
- /** Number of attribute fields truncated to fit the per-attribute cap. */
1461
- truncated_attribute_count: number;
1462
- }
1463
- interface SpanMatchRecord {
1464
- trace_id: string;
1465
- span_id: string;
1466
- span_name: string;
1467
- span_kind: TraceAnalystSpanKind;
1468
- /** JSON pointer-style path to the matched value, e.g.
1469
- * `attributes."llm.input_messages"[2].content`. */
1470
- attribute_path: string;
1471
- matched_text: string;
1472
- context_before: string;
1473
- context_after: string;
1474
- match_offset: number;
1475
- }
1476
- interface SearchTraceResult {
1477
- trace_id: string;
1478
- hits: SpanMatchRecord[];
1479
- total_matches: number;
1480
- has_more: boolean;
1481
- }
1482
- interface SearchSpanResult {
1483
- trace_id: string;
1484
- span_id: string;
1485
- hits: SpanMatchRecord[];
1486
- total_matches: number;
1487
- has_more: boolean;
1488
- }
1489
-
1490
- /**
1491
- * `TraceAnalysisStore` — read-side interface the trace-analyst calls
1492
- * through. Six operations, all bounded:
1493
- *
1494
- * - `getOverview(filters?)` — dataset rollup + sample trace ids.
1495
- * - `queryTraces(filters?, limit, offset)` — paginated summaries.
1496
- * - `countTraces(filters?)` — cheap count without materialisation.
1497
- * - `viewTrace(trace_id, perAttrCap)` — full span list, oversized → summary.
1498
- * - `viewSpans(trace_id, span_ids, perAttrCap)` — surgical span fetch.
1499
- * - `searchTrace(trace_id, regex, max_matches)` — bounded regex hits.
1500
- * - `searchSpan(trace_id, span_id, regex, max_matches)` — single-span search.
1501
- *
1502
- * Multiple implementations ship in the core (`OtlpFileTraceStore`).
1503
- * Downstream callers can supply their own — e.g. a DuckDB-backed
1504
- * adapter or an in-memory adapter for tests — by implementing this
1505
- * interface.
1506
- *
1507
- * Filters compose with AND semantics. Empty/undefined fields impose
1508
- * no constraint. `regex_pattern` is the only opt-in raw-bytes scan —
1509
- * implementations may skip it via `count`/`overview` when not set.
1510
- */
1511
-
1512
- interface TraceAnalysisStore {
1513
- getOverview(filters?: TraceAnalystFilters): Promise<DatasetOverview>;
1514
- queryTraces(opts: {
1515
- filters?: TraceAnalystFilters;
1516
- limit: number;
1517
- offset?: number;
1518
- }): Promise<QueryTracesPage>;
1519
- countTraces(filters?: TraceAnalystFilters): Promise<number>;
1520
- viewTrace(opts: {
1521
- trace_id: string;
1522
- /** Override per-attribute byte cap. Defaults to discovery budget. */
1523
- per_attribute_byte_cap?: number;
1524
- }): Promise<ViewTraceResult>;
1525
- viewSpans(opts: {
1526
- trace_id: string;
1527
- span_ids: readonly string[];
1528
- /** Override per-attribute byte cap. Defaults to surgical budget. */
1529
- per_attribute_byte_cap?: number;
1530
- }): Promise<ViewSpansResult>;
1531
- searchTrace(opts: {
1532
- trace_id: string;
1533
- regex_pattern: string;
1534
- /** Hard cap on matches returned. Default 50. */
1535
- max_matches?: number;
1536
- }): Promise<SearchTraceResult>;
1537
- searchSpan(opts: {
1538
- trace_id: string;
1539
- span_id: string;
1540
- regex_pattern: string;
1541
- max_matches?: number;
1542
- }): Promise<SearchSpanResult>;
1543
- }
1544
-
1545
- /**
1546
- * ChatClient — the single LLM abstraction analysts call.
1547
- *
1548
- * agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
1549
- * graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
1550
- * mixed patterns force every analyst author to pick a transport, which
1551
- * couples analyst code to runtime concerns (cli-bridge vs router vs
1552
- * sandbox-sdk) it shouldn't know about.
1553
- *
1554
- * `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
1555
- * The operator decides at the registry boundary which transport binds
1556
- * to it. Analyst code stays transport-agnostic; swapping production
1557
- * (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
1558
- * line factory call.
1559
- *
1560
- * Designed to coexist: existing `LlmClient` callers and existing
1561
- * `TCloud`-based judges keep working untouched. New analyst code uses
1562
- * `ChatClient`. When old call sites migrate, they pick up budgeting,
1563
- * cancellation, and unified telemetry for free.
1564
- */
1565
-
1566
- /**
1567
- * Unified chat interface. Mirrors LlmCallRequest/Result so the OpenAI-
1568
- * compatible mental model stays. Two methods: a one-shot `chat()` and
1569
- * an `streamChat()` for future agentic loops (not yet exposed).
1570
- */
1571
- interface ChatClient {
1572
- /** Display name of the bound transport — included in telemetry. */
1573
- readonly transport: ChatTransport;
1574
- /** Default model when caller omits — operators bind this per environment. */
1575
- readonly defaultModel?: string;
1576
- /** Total provider attempts this transport can make for one chat call. */
1577
- readonly maximumAttempts?: number;
1578
- /** Implementations must enforce `req.maxTokens` when it is present. */
1579
- chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
1580
- }
1581
- type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
1582
- interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
1583
- /** Optional — falls back to ChatClient.defaultModel. */
1584
- model?: string;
1585
- }
1586
- type ChatResponse = LlmCallResult;
1587
- interface ChatCallOpts {
1588
- /** Cancel the in-flight request. */
1589
- signal?: AbortSignal;
1590
- /** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
1591
- maxCostUsd?: number;
1592
- /** Correlation tag carried into request headers when the transport allows. */
1593
- correlationId?: string;
1594
- /** Stable provider idempotency key for retries/redrives of one paid call. */
1595
- idempotencyKey?: string;
1596
- }
1597
- type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
1598
- interface BaseTransportOpts {
1599
- defaultModel?: string;
1600
- /** Total provider attempts. Required for opaque transports used in capped runs. */
1601
- maximumAttempts?: number;
1602
- }
1603
- interface RouterTransportOpts extends BaseTransportOpts {
1604
- transport: 'router';
1605
- baseUrl?: string;
1606
- apiKey: string;
1607
- }
1608
- interface CliBridgeTransportOpts extends BaseTransportOpts {
1609
- transport: 'cli-bridge';
1610
- baseUrl?: string;
1611
- bearer?: string;
1612
- }
1613
- interface DirectProviderTransportOpts extends BaseTransportOpts {
1614
- transport: 'direct-provider';
1615
- baseUrl: string;
1616
- apiKey: string;
1617
- }
1618
- /**
1619
- * Sandbox-SDK transport. Provided as a thin pass-through: the caller
1620
- * supplies a callable that mimics LlmClient.chat() against an already-
1621
- * configured Sandbox handle. We don't import the SDK here to keep
1622
- * agent-eval dep-free of @tangle-network/sandbox.
1623
- */
1624
- interface SandboxSdkTransportOpts extends BaseTransportOpts {
1625
- transport: 'sandbox-sdk';
1626
- chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
1627
- }
1628
- /**
1629
- * Mock transport for tests. The handler receives the request and returns
1630
- * whatever the test wants. No retries, no JSON-schema degrade.
1631
- */
1632
- interface MockTransportOpts extends BaseTransportOpts {
1633
- transport: 'mock';
1634
- handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
1635
- }
1636
- /**
1637
- * Build a ChatClient bound to a specific transport. The returned client
1638
- * is safe to share across analysts in a single registry run.
1639
- */
1640
- declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
1641
-
1642
- /**
1643
- * Analyst contract — the missing orchestration layer over agent-eval's
1644
- * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
1645
- * SemanticConceptJudge, JudgeFn, ...).
1646
- *
1647
- * Each existing primitive returns its own output shape. The Analyst
1648
- * contract is the single envelope every primitive lifts into, so a
1649
- * registry can run N analysts against a run and a single renderer can
1650
- * compose findings without knowing which analyzer produced them.
1651
- *
1652
- * The contract is intentionally domain-agnostic: nothing here knows
1653
- * about code, voice, RAG, or any particular agent stack. Analysts
1654
- * declare what INPUT KIND they need (a trace store, an artifact dir,
1655
- * a RunRecord, a JudgeInput, or `custom`), and the registry routes
1656
- * the matching input from `AnalystRunInputs`.
1657
- */
1658
-
1659
- /**
1660
- * Unified envelope every analyst emits. Schema-versioned so renderers
1661
- * and time-series diffs survive future field additions.
1662
- */
1663
- interface AnalystFinding {
1664
- schema_version: '1.0.0';
1665
- /**
1666
- * Stable hash over identity-defining fields (analyst_id + canonical
1667
- * claim + area + optional subject). Two findings from two runs that
1668
- * "are the same finding" share this id — that's what `diffFindings`
1669
- * uses to compute appeared/disappeared sets across runs.
1670
- */
1671
- finding_id: string;
1672
- analyst_id: string;
1673
- produced_at: string;
1674
- severity: AnalystSeverity;
1675
- /**
1676
- * Coarse classification. Renderers group by this. Free-form so
1677
- * domain-specific analysts can introduce categories without a
1678
- * schema change ('agent-reasoning', 'verification', 'cost',
1679
- * 'tool-use', 'safety', 'latency', 'data-quality', ...).
1680
- */
1681
- area: string;
1682
- claim: string;
1683
- rationale?: string;
1684
- evidence_refs: EvidenceRef[];
1685
- recommended_action?: string;
1686
- validation_plan?: string;
1687
- /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */
1688
- confidence: number;
1689
- /**
1690
- * Optional subject the finding is about — leaf id, agent id, request
1691
- * id. Included in finding_id when present so per-subject findings
1692
- * diff cleanly across runs.
1693
- */
1694
- subject?: string;
1695
- /** FIREWALL provenance (docs/learning-flywheel.md): true iff this finding was
1696
- * lifted from a JUDGE verdict (an acceptance score), not OBSERVED from the
1697
- * agent's behavior. A judge-derived finding must NEVER be admitted as a
1698
- * steering input — that is the held-out judge leaking into the loop. Set at
1699
- * the lift site (createJudgeAdapter); checked by `assertNoJudgeVerdict`.
1700
- * Provenance, not evidence presence, is the correct discriminator: an
1701
- * evidence-less trace-analyst observation legitimately steers, while a judge
1702
- * verdict that happens to cite an artifact must not. */
1703
- derived_from_judge?: boolean;
1704
- /** Analyst-private extras; renderers ignore unless they know the analyst. */
1705
- metadata?: Record<string, unknown>;
1706
- }
1707
- type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info';
1708
- interface EvidenceRef {
1709
- /**
1710
- * Where the evidence lives. `span` and `event` refer to OTLP trace
1711
- * elements; `artifact` to a file inside the run's artifact tree;
1712
- * `finding` to another AnalystFinding (cross-analyst chaining);
1713
- * `metric` to a named scalar reading the renderer knows how to read.
1714
- */
1715
- kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
1716
- uri: string;
1717
- excerpt?: string;
1718
- }
1719
- /**
1720
- * The discriminator the registry uses to pass the right input.
1721
- * `custom` is the escape hatch — analysts that need something else
1722
- * (e.g. an embedding cache, a partner SDK handle) read it from
1723
- * `AnalystRunInputs.custom[<analyst id>]`.
1724
- */
1725
- type AnalystInputKind = 'trace-store' | 'artifact-dir' | 'run-record' | 'judge-input' | 'custom';
1726
- interface AnalystCost {
1727
- /** `deterministic` analysts MUST NOT call the LLM. */
1728
- kind: 'deterministic' | 'llm';
1729
- /** Optional declared upper bound; the registry can enforce a budget. */
1730
- est_usd_per_run?: number;
1731
- /** Models the analyst expects to use (informational). */
1732
- models?: string[];
1733
- /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */
1734
- settlement_timeout_ms?: number;
1735
- }
1736
- interface AnalystRequirements {
1737
- /** Min number of shots / samples the analyst needs to produce signal. */
1738
- min_shots?: number;
1739
- /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */
1740
- capabilities?: string[];
1741
- }
1742
- /**
1743
- * What's passed to every analyst call. The registry resolves which
1744
- * field the analyst's `inputKind` selects and asserts it's present.
1745
- */
1746
- interface AnalystRunInputs {
1747
- traceStore?: TraceAnalysisStore;
1748
- artifactDir?: string;
1749
- runRecord?: RunRecord;
1750
- judgeInput?: JudgeInput;
1751
- /** Keyed by analyst id; populated by callers that registered custom analysts. */
1752
- custom?: Record<string, unknown>;
1753
- }
1754
- interface AnalystContext {
1755
- runId: string;
1756
- /** Stable correlation id so logs from a single registry.run() share a tag. */
1757
- correlationId: string;
1758
- /** Enforced wall-clock deadline (epoch ms). */
1759
- deadlineMs?: number;
1760
- /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */
1761
- budgetUsd?: number;
1762
- /** Shared paid-call account when the analyst runs inside a larger campaign. */
1763
- costLedger?: CostLedgerHandle;
1764
- /** Attribution phase used when writing to the shared paid-call account. */
1765
- costPhase?: string;
1766
- /**
1767
- * Shared chat client. Analysts that call an LLM go through this so
1768
- * the operator picks transport (sandbox-sdk | router | cli-bridge |
1769
- * direct-provider | mock) at the registry boundary without touching
1770
- * analyst code.
1771
- */
1772
- chat?: ChatClient;
1773
- /**
1774
- * Findings from a prior run the operator wants the analyst to see as
1775
- * retrieval context. Kinds that take advantage of cross-run memory
1776
- * (failure-mode "I saw this cluster last run", knowledge-gap "the wiki
1777
- * page I asked for is still missing") render these into the actor's
1778
- * working set. Filtering is the operator's job: pass the slice that
1779
- * matches the analyst's id, or pass everything and let the kind
1780
- * filter. Empty / absent means no cross-run context.
1781
- */
1782
- priorFindings?: ReadonlyArray<AnalystFinding>;
1783
- /**
1784
- * Findings emitted by analysts that completed earlier in this registry run.
1785
- * This is separate from `priorFindings`: upstream findings are dependency
1786
- * context for the current pass, while prior findings are cross-run memory.
1787
- * The registry populates this only when `RegistryRunOpts.chainFindings` is on.
1788
- */
1789
- upstreamFindings?: ReadonlyArray<AnalystFinding>;
1790
- /**
1791
- * Report metered work independently of findings. This keeps an empty finding
1792
- * set from erasing token/cost telemetry. Multiple receipts are accumulated.
1793
- */
1794
- recordUsage?: (receipt: AnalystUsageReceipt) => void;
1795
- /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */
1796
- tags?: Record<string, string>;
1797
- /** Logger callback — analysts SHOULD prefer this over console.* for testability. */
1798
- log?: (msg: string, fields?: Record<string, unknown>) => void;
1799
- /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
1800
- signal?: AbortSignal;
1801
- }
1802
- /**
1803
- * The minimal contract. Concrete analysts can refine `TInput` so
1804
- * implementations stay type-safe (e.g. a trace analyst's `TInput` is
1805
- * `TraceAnalysisStore`); the registry passes the right field from
1806
- * `AnalystRunInputs` based on `inputKind`.
1807
- */
1808
- interface Analyst<TInput = unknown> {
1809
- /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */
1810
- readonly id: string;
1811
- /** Human-readable. One sentence. */
1812
- readonly description: string;
1813
- readonly inputKind: AnalystInputKind;
1814
- readonly cost: AnalystCost;
1815
- readonly requires?: AnalystRequirements;
1816
- /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */
1817
- readonly version: string;
1818
- analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>;
1819
- }
1820
- /** Metered work performed by one analyst call. */
1821
- interface AnalystUsageReceipt {
1822
- /** Number of model-usage records observed at the provider boundary. */
1823
- calls: number | null;
1824
- /** Null when the provider did not return token accounting. */
1825
- tokens: RunTokenUsage | null;
1826
- /** Observed, estimated, or explicitly uncaptured dollar cost. */
1827
- cost: RunCostProvenance;
1828
- /** Known lower bound when one or more calls have uncaptured cost. */
1829
- knownCostUsd?: number;
1830
- }
1831
- /**
1832
- * Compute the stable finding_id from the identity-defining fields.
1833
- * Default implementation hashes {analyst_id, area, subject, normalized claim}.
1834
- * Analysts that emit findings whose claim text varies per run (timestamps,
1835
- * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
1836
- * or (b) move the variable part into `rationale`/`metadata` and keep the
1837
- * `claim` static.
1838
- */
1839
- declare function computeFindingId(input: {
1840
- analyst_id: string;
1841
- area: string;
1842
- subject?: string;
1843
- claim: string;
1844
- /** Override the claim for hashing — use when the displayed claim has run-specific bits. */
1845
- id_basis?: string;
1846
- }): string;
1847
- /**
1848
- * Convenience factory: produce a fully-formed AnalystFinding with the
1849
- * id computed automatically. Analyst code stays terse.
1850
- */
1851
- declare function makeFinding(init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {
1852
- id_basis?: string;
1853
- produced_at?: string;
1854
- }): AnalystFinding;
1855
- interface AnalystRunSummary {
1856
- analyst_id: string;
1857
- status: 'ok' | 'skipped' | 'failed';
1858
- /** Why skipped — missing input, budget exceeded, capability unmet. */
1859
- reason?: string;
1860
- findings_count: number;
1861
- latency_ms: number;
1862
- cost_usd: number;
1863
- /**
1864
- * Additive receipt for model usage. Registry-produced summaries populate it
1865
- * even when the analyst emits no findings. `cost_usd` remains the legacy
1866
- * numeric field; inspect `usage.cost` before treating zero as observed.
1867
- */
1868
- usage?: AnalystUsageReceipt;
1869
- /** When `status='failed'`: the error class + message, never the full stack. */
1870
- error?: {
1871
- class: string;
1872
- message: string;
1873
- };
1874
- }
1875
- interface AnalystRunResult {
1876
- run_id: string;
1877
- correlation_id: string;
1878
- started_at: string;
1879
- ended_at: string;
1880
- findings: AnalystFinding[];
1881
- per_analyst: AnalystRunSummary[];
1882
- /** Total LLM cost in USD across all analysts in this registry.run(). */
1883
- total_cost_usd: number;
1884
- /**
1885
- * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only
1886
- * the known subtotal and must not be treated as the run's total spend.
1887
- */
1888
- total_cost_provenance?: RunCostProvenance;
1889
- }
1890
- /**
1891
- * Events emitted by `AnalystRegistry.runStream(...)` in real time as
1892
- * the registry executes. UIs subscribe via `for await (const ev of
1893
- * registry.runStream(...))`; `registry.run(...)` is a thin collector
1894
- * over the same stream, so the two surfaces share their invariants.
1895
- *
1896
- * Per-finding events are intentionally omitted — analyzers are batch
1897
- * operations (an Ax actor returns the full `findings:json[]` at the
1898
- * end of the responder), so streaming inside one analyst would only
1899
- * emit partial JSON consumers can't render. The kind-completion event
1900
- * is the right granularity; subscribers wanting per-finding rendering
1901
- * iterate `event.findings` themselves.
1902
- */
1903
- type AnalystRunEvent = {
1904
- type: 'run-started';
1905
- run_id: string;
1906
- correlation_id: string;
1907
- started_at: string;
1908
- /** The ordered list of analyst ids the registry will run. */
1909
- analyst_ids: ReadonlyArray<string>;
1910
- } | {
1911
- type: 'analyst-skipped';
1912
- summary: AnalystRunSummary;
1913
- } | {
1914
- type: 'analyst-started';
1915
- analyst_id: string;
1916
- started_at: string;
1917
- } | {
1918
- type: 'analyst-completed';
1919
- /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */
1920
- summary: AnalystRunSummary;
1921
- findings: ReadonlyArray<AnalystFinding>;
1922
- } | {
1923
- type: 'run-completed';
1924
- result: AnalystRunResult;
1925
- };
1926
-
1927
- /**
1928
- * Adapter factories — lift each existing agent-eval primitive into the
1929
- * Analyst contract without re-implementing it.
1930
- *
1931
- * Five primitives, five factories. Each one:
1932
- * - Builds an Analyst with a stable id (caller chooses; defaults
1933
- * given), a sensible default `inputKind`, a version derived from
1934
- * the wrapped primitive's version + an adapter revision, and an
1935
- * `analyze()` that calls the primitive and lifts its output to
1936
- * AnalystFinding[] using `makeFinding()`.
1937
- * - Maps severities: the existing `Severity` ('critical' | 'major' |
1938
- * 'minor' | 'info') projects onto AnalystSeverity ('critical' |
1939
- * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →
1940
- * 'medium'. Domain analysts that want finer-grained mapping override.
1941
- *
1942
- * Adapters never own state. Calling the same factory twice with the
1943
- * same primitive instance is safe.
1944
- */
1945
-
1
+ import { a as MultiLayerVerifier, l as VerifyOptions, o as Severity } from "../multi-layer-verifier-BHY1gWAc.js";
2
+ import { C as FindingSubjectKind, D as parseFindingSubject, E as findingSubjectGrammarPromptFor, F as createAnalystAi, H as SemanticConceptJudgeInput, O as renderFindingSubject, P as CreateAnalystAiConfig, S as FindingSubject, T as KIND_EXPECTED_SUBJECTS, U as SemanticConceptJudgeOptions, Y as RunTrace, _ as defaultIsMaterial, a as SkillUsageScanConfig, b as FINDING_SUBJECT_KINDS, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as PersistedFinding, h as FindingsStore, i as SkillUsageReport, k as BehavioralMetrics, l as KNOWLEDGE_POISONING_KIND_SPEC, m as FindingsDiff, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as DiffPolicy, q as RunCritic, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as diffFindings, w as FindingSubjectStringSchema, x as FINDING_SUBJECT_SYNTAX, y as FINDING_SUBJECT_GRAMMAR_PROMPT } from "../skill-usage-C_pXm7lP.js";
3
+ import { c as CostLedgerHandle } from "../cost-ledger-Dye6jCgg.js";
4
+ import { o as LlmClientOptions } from "../llm-client-B_nIBlYo.js";
5
+ import { A as ChatClient, B as SandboxSdkTransportOpts, F as CreateChatClientOpts, I as CustomTransportOpts, L as DirectProviderTransportOpts, M as ChatResponse, N as ChatTransport, P as CliBridgeTransportOpts, R as MockTransportOpts, V as createChatClient, j as ChatRequest, k as ChatCallOpts, m as JudgeInput, p as JudgeFn, z as RouterTransportOpts } from "../types-DGsxbAEd.js";
6
+ import { t as TraceAnalysisStore } from "../store-CxJry_cs.js";
7
+ import { A as AnalystRunResult, C as AnalystContext, D as AnalystRequirements, E as AnalystInputKind, F as computeFindingId, I as makeFinding, M as AnalystSeverity, N as AnalystUsageReceipt, O as AnalystRunEvent, P as EvidenceRef, S as Analyst, T as AnalystFinding, _ as RawAnalystEvidenceSchema, a as AnalystRegistryOptions, b as evidenceRefsFromRawFinding, c as CreateTraceAnalystKindOpts, d as createTraceAnalystKind, f as renderPriorFindings, g as RawAnalystEvidence, h as RAW_FINDING_SCHEMA_PROMPT, i as AnalystRegistry, j as AnalystRunSummary, k as AnalystRunInputs, l as TraceAnalystGolden, m as ANALYST_SEVERITIES, n as buildDefaultAnalystRegistry, o as BudgetPolicy, p as renderUpstreamFindings, r as AnalystHooks, s as RegistryRunOpts, t as DefaultAnalystRegistryOptions, u as TraceAnalystKindSpec, v as RawAnalystFinding, w as AnalystCost, x as parseRawFinding, y as RawAnalystFindingSchema } from "../default-registry-CNPo-Vsb.js";
8
+ import { AxFunction } from "@ax-llm/ax";
9
+ //#region src/analyst/adapters.d.ts
1946
10
  declare function liftSeverity(s: Severity): AnalystSeverity;
1947
11
  interface VerifierAdapterOpts<Env> {
1948
- id?: string;
1949
- area?: string;
1950
- verifier: MultiLayerVerifier<Env>;
1951
- /**
1952
- * The verifier expects an `env` per run. Adapters take it from
1953
- * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.
1954
- */
1955
- options?: Omit<VerifyOptions<Env>, 'env'>;
12
+ id?: string;
13
+ area?: string;
14
+ verifier: MultiLayerVerifier<Env>;
15
+ /**
16
+ * The verifier expects an `env` per run. Adapters take it from
17
+ * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.
18
+ */
19
+ options?: Omit<VerifyOptions<Env>, 'env'>;
1956
20
  }
1957
21
  declare function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env>;
1958
22
  interface RunCriticAdapterOpts {
1959
- id?: string;
1960
- area?: string;
1961
- critic?: RunCritic;
1962
- /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */
1963
- threshold?: number;
23
+ id?: string;
24
+ area?: string;
25
+ critic?: RunCritic;
26
+ /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */
27
+ threshold?: number;
1964
28
  }
1965
29
  declare function createRunCriticAdapter(opts?: RunCriticAdapterOpts): Analyst<RunTrace>;
1966
30
  interface JudgeAdapterOpts {
1967
- id?: string;
1968
- area?: string;
1969
- judge: JudgeFn;
1970
- /** TCloud handle the JudgeFn calls. */
1971
- tcloud: TCloud;
1972
- /** Optional cost classification — most judges call an LLM. */
1973
- cost?: Analyst['cost'];
1974
- /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */
1975
- threshold?: number;
31
+ id?: string;
32
+ area?: string;
33
+ judge: JudgeFn;
34
+ /** Chat client passed to the JudgeFn. */
35
+ chat: ChatClient;
36
+ /** Optional cost classification — most judges call an LLM. */
37
+ cost?: Analyst['cost'];
38
+ /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */
39
+ threshold?: number;
1976
40
  }
1977
41
  declare function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput>;
1978
42
  interface SemanticConceptJudgeAdapterOpts {
1979
- id?: string;
1980
- area?: string;
1981
- /** Registry context owns cancellation and the per-analyst cost ledger. */
1982
- options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>;
1983
- /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */
1984
- settlementTimeoutMs?: number;
43
+ id?: string;
44
+ area?: string;
45
+ /** Registry context owns cancellation and the per-analyst cost ledger. */
46
+ options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>;
47
+ /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */
48
+ settlementTimeoutMs?: number;
1985
49
  }
1986
50
  declare function createSemanticConceptJudgeAdapter(opts?: SemanticConceptJudgeAdapterOpts): Analyst<SemanticConceptJudgeInput>;
1987
-
1988
- interface CreateAnalystAiConfig {
1989
- /** OpenAI-compatible API key forwarded as `Authorization: Bearer`.
1990
- * cli-bridge ignores the value on loopback but Ax requires a non-empty string. */
1991
- apiKey: string;
1992
- /** OpenAI-compatible base URL — e.g. `https://router.tangle.tools/v1` or a
1993
- * cli-bridge loopback. */
1994
- baseUrl?: string;
1995
- /** Additional headers required by the gateway, such as tenant or execution policy. */
1996
- headers?: Record<string, string>;
1997
- /** Model id forwarded to analyst calls. */
1998
- model: string;
1999
- /** Ax provider name. Defaults to the OpenAI-compatible client. */
2000
- provider?: AxAIArgs<unknown>['name'];
2001
- }
2002
- /**
2003
- * Construct the `AxAIService` an analyst kind calls through
2004
- * (`createTraceAnalystKind({ ai })`).
2005
- *
2006
- * Ax's `ai()` pins `config.model` to the OpenAI catalog enum, but every
2007
- * OpenAI-compatible router an analyst points at (router.tangle.tools,
2008
- * cli-bridge) accepts arbitrary model ids (claude-code/sonnet, openai/gpt-5.4,
2009
- * …). Consumers were each re-rolling `ai({ name, apiKey, apiURL, config })`
2010
- * behind an `as (a: any) => any` cast to dodge the enum; this is the one
2011
- * canonical constructor so they don't have to — and don't take a direct
2012
- * `@ax-llm/ax` dependency for it.
2013
- */
2014
- declare function createAnalystAi(config: CreateAnalystAiConfig): AxAIService;
2015
-
2016
- /**
2017
- * Deterministic behavioral metrics over OTLP spans — pure arithmetic, no LLM.
2018
- *
2019
- * These are the model-independent multiplier: the four trace-quality signals a
2020
- * tolerant analyzer (e.g. HALO) re-derives per run inside the model — token
2021
- * growth, output decay, tool monoculture, missing self-verification — computed
2022
- * here once, in TypeScript, with zero model judgment. A finding that falls out
2023
- * of arithmetic is trivially model-agnostic and cannot hallucinate the trend.
2024
- *
2025
- * General, not trace-specific: the detectors key off token trajectories and
2026
- * tool usage present in any agentic OTLP trace, not any one benchmark.
2027
- */
2028
-
2029
- type SuboptimalCode = 'monotonic-input-growth' | 'output-length-decay' | 'single-tool-dependency' | 'no-self-verification';
2030
- interface SuboptimalSignal {
2031
- code: SuboptimalCode;
2032
- severity: 'high' | 'medium' | 'low';
2033
- /** Human-readable claim, with the backing numbers inlined. */
2034
- detail: string;
2035
- /** The exact figures the detector fired on — auditable, no model in the loop. */
2036
- evidence: Record<string, number | string | boolean>;
2037
- }
2038
- interface BehavioralMetrics {
2039
- /** The only trace represented by these metrics; null when spans are empty. */
2040
- traceId: string | null;
2041
- llmCallCount: number;
2042
- /** Causally serial LLM timelines. Parallel branches are never joined. */
2043
- tokenSequences: BehavioralTokenSequence[];
2044
- /** Token values from the longest serial timeline, retained for convenience. */
2045
- inputTokenTrajectory: number[];
2046
- outputTokenTrajectory: number[];
2047
- toolHistogram: Record<string, number>;
2048
- totalToolCalls: number;
2049
- distinctTools: number;
2050
- /** distinct/total tool calls; 1.0 when there are no tool calls. */
2051
- toolDiversityRatio: number;
2052
- hasSelfVerification: boolean;
2053
- signals: SuboptimalSignal[];
2054
- }
2055
- interface BehavioralTokenSequence {
2056
- scopeId: string;
2057
- spanIds: string[];
2058
- inputTokenTrajectory: Array<number | null>;
2059
- outputTokenTrajectory: Array<number | null>;
2060
- }
2061
-
2062
- /**
2063
- * `behavioralAnalyst` — a DETERMINISTIC analyst (cost.kind = 'deterministic',
2064
- * never calls the LLM). It produces the efficiency/behavioral findings a
2065
- * tolerant agentic analyzer (HALO) re-derives per run inside the model —
2066
- * context bloat, output decay, tool monoculture, missing self-verification —
2067
- * directly from arithmetic over spans (`computeTraceMetrics`).
2068
- *
2069
- * Why it matters: these findings are model-agnostic BY CONSTRUCTION (no model
2070
- * in the loop), so they cannot return 0 on a weak model the way the Ax-RLM
2071
- * does — and they are strictly more reliable than HALO, which spends tokens
2072
- * re-deriving the same numbers and can hallucinate the trend. The agentic
2073
- * RLM kinds remain for SEMANTIC findings that genuinely need a model; this
2074
- * analyst owns the behavioral class.
2075
- */
2076
-
51
+ //#endregion
52
+ //#region src/analyst/behavioral-analyst.d.ts
2077
53
  /**
2078
54
  * Map computed signals → structured AnalystFindings. Pure: no LLM, no clock
2079
55
  * dependence beyond `produced_at` (overridable for deterministic tests).
2080
56
  */
2081
57
  declare function deriveEfficiencyFindings(metrics: BehavioralMetrics, opts?: {
2082
- analystId?: string;
2083
- producedAt?: string;
58
+ analystId?: string;
59
+ producedAt?: string;
2084
60
  }): AnalystFinding[];
2085
61
  /** The deterministic behavioral/efficiency analyst (no LLM, any-model). */
2086
62
  declare function behavioralAnalyst(): Analyst<TraceAnalysisStore>;
2087
-
2088
- /**
2089
- * Typed Ax output for analyst findings.
2090
- *
2091
- * Replaces the legacy `findings:string[]` pattern (where every bullet
2092
- * became a flat-severity `AnalystFinding`) with a structured object
2093
- * array. Ax binds the field as `findings:json[]` so the provider emits
2094
- * native structured output; at the kind-factory boundary we Zod-validate
2095
- * each emitted finding so malformed rows fail loud instead of being
2096
- * silently lifted with default severity.
2097
- *
2098
- * Why not `f.object().array()` directly in the signature? The Ax
2099
- * signature string `question:string -> findings:json[]` already lets
2100
- * the provider emit JSON arrays. A Zod boundary is required either
2101
- * way (the provider can return any JSON), and Zod gives us a single
2102
- * validation surface independent of which Ax version is installed.
2103
- */
2104
-
2105
- declare const ANALYST_SEVERITIES: readonly ["critical", "high", "medium", "low", "info"];
2106
- declare const RawAnalystEvidenceSchema: z.ZodObject<{
2107
- uri: z.ZodString;
2108
- excerpt: z.ZodOptional<z.ZodString>;
2109
- }, z.core.$strict>;
2110
- type RawAnalystEvidence = z.infer<typeof RawAnalystEvidenceSchema>;
2111
- /** Original public schema retained for stored rows and callback contracts. */
2112
- declare const RawAnalystFindingSchema: z.ZodObject<{
2113
- evidence_uri: z.ZodString;
2114
- evidence_excerpt: z.ZodOptional<z.ZodString>;
2115
- severity: z.ZodEnum<{
2116
- critical: "critical";
2117
- info: "info";
2118
- low: "low";
2119
- high: "high";
2120
- medium: "medium";
2121
- }>;
2122
- claim: z.ZodString;
2123
- subject: z.ZodOptional<z.ZodString>;
2124
- confidence: z.ZodNumber;
2125
- rationale: z.ZodOptional<z.ZodString>;
2126
- recommended_action: z.ZodOptional<z.ZodString>;
2127
- }, z.core.$strict>;
2128
- type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
2129
- /**
2130
- * Canonical plural-evidence contract. The preprocessor accepts the original
2131
- * `evidence_uri` / `evidence_excerpt` pair and normalizes it into one evidence
2132
- * item so persisted rows and older model fixtures remain readable. New output
2133
- * always receives the plural shape.
2134
- */
2135
- declare const CanonicalRawAnalystFindingSchema: z.ZodPreprocess<z.ZodObject<{
2136
- evidence: z.ZodArray<z.ZodObject<{
2137
- uri: z.ZodString;
2138
- excerpt: z.ZodOptional<z.ZodString>;
2139
- }, z.core.$strict>>;
2140
- severity: z.ZodEnum<{
2141
- critical: "critical";
2142
- info: "info";
2143
- low: "low";
2144
- high: "high";
2145
- medium: "medium";
2146
- }>;
2147
- claim: z.ZodString;
2148
- subject: z.ZodOptional<z.ZodString>;
2149
- confidence: z.ZodNumber;
2150
- rationale: z.ZodOptional<z.ZodString>;
2151
- recommended_action: z.ZodOptional<z.ZodString>;
2152
- }, z.core.$strict>>;
2153
- type CanonicalRawAnalystFinding = z.infer<typeof CanonicalRawAnalystFindingSchema>;
2154
- /**
2155
- * Description embedded into the actor prompt so the LLM knows what
2156
- * shape to emit. Kept here so kinds share one source of truth rather
2157
- * than restating the schema in every prompt.
2158
- */
2159
- declare const RAW_FINDING_SCHEMA_PROMPT = "Each finding MUST be a strict JSON object with:\n - severity: \"critical\" | \"high\" | \"medium\" | \"low\" | \"info\"\n - claim: one-sentence statement (max 2000 chars)\n - subject?: one exact subject form listed by this kind; omit rather than guess\n - evidence: REQUIRED non-empty array of {\"uri\": string, \"excerpt\"?: string}. Use real identifiers with span://, event://, artifact://, metric://, or finding://. Include a short exact quote in excerpt when available. If nothing is citable, do not emit the finding.\n - confidence: number 0..1 (0.9+ exact evidence; 0.6-0.8 inferred pattern; <0.5 speculative)\n - rationale?: one or two reasoning sentences\n - recommended_action?: concrete imperative change; omit for descriptive findings\n\nUnknown fields are rejected. Do not emit area; the factory assigns it. Emit [] when there are no findings. Never fabricate evidence.";
2160
- /** Convert canonical raw citations into the public finding evidence envelope. */
2161
- declare function evidenceRefsFromRawFinding(finding: CanonicalRawAnalystFinding): EvidenceRef[];
2162
- /**
2163
- * Validate the original singular-evidence shape. This public parser retains
2164
- * its pre-canonicalization result type so existing callback code and stored
2165
- * rows continue to receive exactly the object accepted by
2166
- * {@link RawAnalystFindingSchema}.
2167
- */
2168
- declare function parseRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): RawAnalystFinding | null;
2169
- /** Validate model output and normalize original singular citations. */
2170
- declare function parseCanonicalRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): CanonicalRawAnalystFinding | null;
2171
-
2172
- /**
2173
- * Analyst-kind factory — the typed way to define trace analysts.
2174
- *
2175
- * A "kind" is a specialized analyst whose actor prompt, tool subset,
2176
- * and bounded Ax subqueries target one failure-mode lens (failure-mode
2177
- * classification, knowledge gap discovery, knowledge poisoning,
2178
- * self-improvement, ...). Kinds emit findings in the typed
2179
- * `CanonicalRawAnalystFinding` shape via a JSON-array Ax output; the factory
2180
- * validates each row with Zod and lifts it into `AnalystFinding[]`.
2181
- *
2182
- * Composition rules:
2183
- * - Each kind owns its actor description. No generic "answer this
2184
- * question" prompt — the prompt names the failure lens.
2185
- * - Each kind picks a narrow tool subset from `ANALYST_TOOL_GROUPS`.
2186
- * A kind that never needs full-trace dumps can drop `viewTrace` /
2187
- * `viewSpans` and stay cheap.
2188
- * - Each kind declares its subquery + parallelism budget. Discovery-heavy
2189
- * kinds can fan out more bounded semantic questions than narrow lenses.
2190
- *
2191
- * Optimizer hook: kinds may declare `goldens` — labeled examples used
2192
- * by `AxBootstrapFewShot` / `AxGEPA` to fit the actor
2193
- * description programmatically. Stored on the kind, not the registry,
2194
- * because the right metric is kind-specific.
2195
- */
2196
-
2197
- /**
2198
- * Per-kind specification. The factory turns this into a regular
2199
- * `Analyst<TraceAnalysisStore>` ready for `AnalystRegistry.register()`.
2200
- */
2201
- interface TraceAnalystKindSpec {
2202
- /** Stable id. Appears in finding_id, telemetry, and registry exclusions. */
2203
- id: string;
2204
- /** One-sentence description shown in `registry.list()`. */
2205
- description: string;
2206
- /** Coarse classification stamped on every emitted finding (`failure-mode`, `knowledge-gap`, ...). */
2207
- area: string;
2208
- /** Bump on any breaking change to the actor prompt or output schema. */
2209
- version: string;
2210
- /** Actor system prompt. Must instruct the LLM to emit `findings` per the schema. */
2211
- actorDescription: string;
2212
- /** Tool functions the actor may call. Pick narrow subsets via `ANALYST_TOOL_GROUPS`. */
2213
- buildTools: (store: TraceAnalysisStore) => AxFunction[];
2214
- /** Bounded semantic subqueries. `maxCalls: 0` disables model fan-out. */
2215
- subqueries?: {
2216
- maxCalls: number;
2217
- maxParallel?: number;
2218
- };
2219
- /** Actor turn cap. Default 12. */
2220
- maxTurns?: number;
2221
- /** Runtime char cap. Default 6000. */
2222
- maxRuntimeChars?: number;
2223
- /** Maximum output tokens for every actor and subquery model call. Default 4096. */
2224
- maxOutputTokens?: number;
2225
- /** Cost classification surfaced in `registry.list()` and budget enforcement. */
2226
- cost: AnalystCost;
2227
- /** Per-finding-row hook — kinds may reject / rewrite before lifting. */
2228
- postProcess?: (row: RawAnalystFinding, ctx: AnalystContext) => RawAnalystFinding | null;
2229
- /** Minimum citations per finding. Default 1; rows below it are rejected. */
2230
- minimumEvidenceCitations?: number;
2231
- /** Optional optimizer hook — populated when a kind wants to fit its prompt against labeled examples. */
2232
- goldens?: TraceAnalystGolden[];
2233
- }
2234
- /**
2235
- * One labeled example consumed by Ax optimizers (MIPRO / GEPA / Bootstrap).
2236
- * Each input is the same `{question}` an analyst would receive; `expected`
2237
- * is the ground-truth finding set a fitted prompt should produce on this
2238
- * input. Metric: kind-specific (default: F1 on `finding_id` overlap).
2239
- */
2240
- interface TraceAnalystGolden {
2241
- question: string;
2242
- expected: ReadonlyArray<Omit<CanonicalRawAnalystFinding, 'confidence'>>;
2243
- }
2244
- interface CreateTraceAnalystKindOpts {
2245
- /** AxAIService bound at registration time. */
2246
- ai: AxAIService;
2247
- /** Required unless `ai` was created by {@link createAnalystAi}. */
2248
- model?: string;
2249
- /** Override the spec's `version` (e.g. when an optimizer has fitted a new prompt). */
2250
- versionSuffix?: string;
2251
- /**
2252
- * Optional two-phase recovery: when the agentic harvest is empty but the
2253
- * actor produced a substantive free-form `report`, extract findings from that
2254
- * prose via a tolerant chat-completions pass (`structureFindings`) — no
2255
- * strict-emission contract, so it works on weak models. Omit to leave the
2256
- * actor's harvest as-is (the report is still surfaced fail-loud either way).
2257
- */
2258
- recovery?: {
2259
- baseUrl: string;
2260
- apiKey?: string;
2261
- model?: string;
2262
- fetchImpl?: typeof fetch;
2263
- };
2264
- /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */
2265
- settlementTimeoutMs?: number;
2266
- }
2267
- /**
2268
- * Build an `Analyst<TraceAnalysisStore>` from a kind spec.
2269
- *
2270
- * Lifts the Ax pipeline once at registration time so the registry
2271
- * gets a stateless analyst. The Ax agent is freshly constructed per
2272
- * `analyze()` call (the agent carries chat-log + usage state we don't
2273
- * want shared across analyst runs).
2274
- */
2275
- declare function createTraceAnalystKind(spec: TraceAnalystKindSpec, opts: CreateTraceAnalystKindOpts): Analyst<TraceAnalysisStore>;
2276
- /**
2277
- * Render a compact prior-findings block the actor reads alongside its
2278
- * brief. Each row is one line so the actor can scan dozens cheaply.
2279
- * The kind's prompt instructs the actor to (a) check whether a new
2280
- * cluster matches a prior `finding_id` (carry the id forward via
2281
- * `id_basis` to keep diffs stable) and (b) raise severity / confidence
2282
- * when a prior finding has reappeared without remediation.
2283
- *
2284
- * Returns the empty string when there are no prior findings — most
2285
- * runs are "first-of-its-kind" and the prompt stays unchanged.
2286
- *
2287
- * Exported for tests + for consumers that build their own actor
2288
- * prompts (e.g. specialized analysts living outside the default kinds).
2289
- */
2290
- declare function renderPriorFindings(prior: AnalystContext['priorFindings']): string;
2291
- /** Render findings produced earlier in this same registry run. */
2292
- declare function renderUpstreamFindings(upstream: AnalystContext['upstreamFindings']): string;
2293
-
2294
- /**
2295
- * AnalystRegistry — orchestrate N analysts against one run.
2296
- *
2297
- * Owns three responsibilities and only three:
2298
- * 1. Registration — ids must be unique; bad registrations fail loudly
2299
- * at register-time, not run-time.
2300
- * 2. Routing — each analyst declares its `inputKind`; the registry
2301
- * picks the matching field from AnalystRunInputs and skips the
2302
- * analyst with a logged reason if it's missing.
2303
- * 3. Isolation — one analyst's exception MUST NOT stop other analysts.
2304
- * Failed analysts produce zero findings + a 'failed' summary row.
2305
- *
2306
- * Cross-cutting concerns (telemetry, error → finding conversion, cost
2307
- * ingestion, storage rotation) live in `AnalystHooks`. Budget shaping
2308
- * (equal split vs weighted vs custom) lives in `BudgetPolicy`. Both
2309
- * have sensible defaults; consumers override only what they need.
2310
- */
2311
-
2312
- interface AnalystHooks {
2313
- /** Before analyze() — last chance to mutate ctx (e.g. inject tags, override budget). */
2314
- onBeforeAnalyze?(args: {
2315
- analyst: Analyst;
2316
- ctx: AnalystContext;
2317
- runId: string;
2318
- }): void | Promise<void>;
2319
- /** After every analyst (ok | failed | skipped). Use for telemetry, ingestion, rotation. */
2320
- onAfterAnalyze?(args: {
2321
- analyst: Analyst;
2322
- summary: AnalystRunSummary;
2323
- findings: AnalystFinding[];
2324
- runId: string;
2325
- }): void | Promise<void>;
2326
- /**
2327
- * On analyst exception. Hook MAY return findings to convert the
2328
- * error into structured findings; the summary still reports 'failed'.
2329
- * Return void to keep the default empty-findings behavior.
2330
- */
2331
- onError?(args: {
2332
- analyst: Analyst;
2333
- error: Error;
2334
- runId: string;
2335
- }): AnalystFinding[] | undefined | Promise<AnalystFinding[] | undefined>;
2336
- /** Once after registry.run() completes. Use for final aggregation, persistence. */
2337
- onComplete?(args: {
2338
- result: AnalystRunResult;
2339
- }): void | Promise<void>;
2340
- }
2341
- interface BudgetPolicy {
2342
- /** Overall USD cap across the registry.run(). */
2343
- totalUsd?: number;
2344
- /** Per-analyst weight for the default allocator. Missing ids get weight 1. */
2345
- weights?: Record<string, number>;
2346
- /**
2347
- * Custom allocator — receives the analyst, remaining/total budget, and
2348
- * the count of analysts that will run. Returns the per-analyst budget
2349
- * (or undefined only when the run has no overall cap). Overrides weights
2350
- * when set.
2351
- */
2352
- allocate?: (args: {
2353
- analyst: Analyst;
2354
- totalUsd: number | undefined;
2355
- remainingUsd: number | undefined;
2356
- runningCount: number;
2357
- }) => number | undefined;
2358
- }
2359
- interface AnalystRegistryOptions {
2360
- /** Shared chat client passed to every LLM analyst via AnalystContext. */
2361
- chat?: ChatClient;
2362
- /** Logger callback. Defaults to a no-op. */
2363
- log?: (msg: string, fields?: Record<string, unknown>) => void;
2364
- /** Hooks invoked around analyze() — observability + customization seam. */
2365
- hooks?: AnalystHooks;
2366
- /** Default budget when run() doesn't override. */
2367
- defaultBudget?: BudgetPolicy;
2368
- }
2369
- interface RegistryRunOpts {
2370
- /** Restrict to a subset of registered analysts by id. */
2371
- only?: string[];
2372
- /** Skip these analysts even if registered. Useful for cheap iteration. */
2373
- skip?: string[];
2374
- /** Budget policy — totalUsd + optional weights/allocator. Falls back to options.defaultBudget. */
2375
- budget?: BudgetPolicy;
2376
- /** Active-work cap for the complete registry run. Model receipt settlement may follow. */
2377
- timeoutMs?: number;
2378
- /** Abort signal — forwarded into every analyst's context. */
2379
- signal?: AbortSignal;
2380
- /** Shared paid-call account forwarded to every analyst. */
2381
- costLedger?: CostLedgerHandle;
2382
- /** Attribution phase for calls written to `costLedger`. */
2383
- costPhase?: string;
2384
- /** Tags echoed into AnalystContext.tags — useful for tracking environment/version in findings. */
2385
- tags?: Record<string, string>;
2386
- /**
2387
- * Prior-run findings made available as retrieval context to every
2388
- * analyst via `ctx.priorFindings`. The registry forwards the slice
2389
- * whose `analyst_id` matches each registered analyst so a kind sees
2390
- * only its own history. Pass `{ '*': findings }` to broadcast to
2391
- * every analyst (useful when several kinds share the same historical
2392
- * context). For findings from this run, use `chainFindings` instead.
2393
- */
2394
- priorFindings?: ReadonlyArray<AnalystFinding> | Record<string, ReadonlyArray<AnalystFinding>>;
2395
- /**
2396
- * Pass findings produced earlier in this registry run to each later analyst
2397
- * via `ctx.upstreamFindings`. Registration order is dependency order.
2398
- * Disabled by default because independent analyst suites must opt in.
2399
- */
2400
- chainFindings?: boolean;
2401
- }
2402
- declare class AnalystRegistry {
2403
- private readonly analysts;
2404
- private readonly options;
2405
- constructor(options?: AnalystRegistryOptions);
2406
- register(analyst: Analyst): void;
2407
- list(): ReadonlyArray<{
2408
- id: string;
2409
- description: string;
2410
- version: string;
2411
- cost: Analyst['cost'];
2412
- }>;
2413
- run(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): Promise<AnalystRunResult>;
2414
- /**
2415
- * Streaming counterpart to `run()`. Emits `AnalystRunEvent` values
2416
- * in real time — `run-started`, then per-analyst `skipped` /
2417
- * `started` / `completed`, then a terminal `run-completed` whose
2418
- * payload is the full `AnalystRunResult`. UIs use this to render
2419
- * progress; persistence consumers use `run()` and read the result.
2420
- *
2421
- * Hooks (`onBeforeAnalyze` / `onAfterAnalyze` / `onError` /
2422
- * `onComplete`) fire as before — streaming is additive, not a hook
2423
- * replacement.
2424
- */
2425
- runStream(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): AsyncGenerator<AnalystRunEvent, void, void>;
2426
- private selectAnalysts;
2427
- private routeInput;
2428
- }
2429
-
2430
- /**
2431
- * `buildDefaultAnalystRegistry` — the canonical analyst suite, so consumers
2432
- * stop hand-wiring `new AnalystRegistry()` + per-kind `createTraceAnalystKind`.
2433
- *
2434
- * The deterministic `behavioralAnalyst` is ALWAYS registered (it needs no
2435
- * model and is model-agnostic by construction). The agentic RLM kinds are
2436
- * registered only when an `ai` service is supplied — so a caller with no LLM
2437
- * still gets the full behavioral/efficiency diagnosis, and the substrate's
2438
- * "any model (including no model)" guarantee holds at the suite level.
2439
- */
2440
-
2441
- interface DefaultAnalystRegistryOptions {
2442
- /** Ax service for the agentic RLM kinds. Omit → only the deterministic analyst. */
2443
- ai?: AxAIService;
2444
- /** Required unless `ai` was created by `createAnalystAi`. */
2445
- model?: string;
2446
- /** Which agentic kinds to register when `ai` is present. Default = the shipped suite. */
2447
- kinds?: readonly TraceAnalystKindSpec[];
2448
- /** Set false to omit the deterministic behavioral analyst (default: include). */
2449
- includeBehavioral?: boolean;
2450
- /** Forwarded to the AnalystRegistry constructor (signal, tags, priorFindings). */
2451
- registry?: AnalystRegistryOptions;
2452
- }
2453
- declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
2454
-
2455
- /**
2456
- * Typed `FindingSubject` — the canonical grammar every analyst kind emits.
2457
- *
2458
- * Background: kind actor prompts have always documented a subject grammar
2459
- * (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`) but the
2460
- * LLM was unconstrained — it could emit `subject: "fix the prompt"`
2461
- * (prose) and downstream adapters routed on `startsWith(...)` would
2462
- * silently skip it. Every per-vertical `ImprovementAdapter` had a
2463
- * routing table that mostly caught nothing.
2464
- *
2465
- * This module fixes that:
2466
- * - `parseFindingSubject(raw)` — returns the typed `FindingSubject`
2467
- * when `raw` matches the grammar, else `null`. Used at the
2468
- * `RawAnalystFindingSchema` boundary so malformed subjects are
2469
- * rejected loudly instead of silently lifted into the registry.
2470
- * - `FindingSubjectKind` — the union of valid locus categories. Each
2471
- * variant carries the typed components downstream adapters resolve
2472
- * against the agent's surface manifest (no string parsing in the
2473
- * adapter).
2474
- * - `FINDING_SUBJECT_GRAMMAR_PROMPT` — single source of truth for the
2475
- * grammar string embedded in kind actor prompts. Drift between
2476
- * prompt and parser is impossible if every kind imports this.
2477
- *
2478
- * The grammar is intentionally NARROW — only loci the substrate's
2479
- * default `ImprovementAdapter` / `KnowledgeAdapter` can act on. A
2480
- * finding with a subject outside this set fails the parser; the kind
2481
- * author either extends the grammar here (and adds adapter routing)
2482
- * or rephrases the prompt to map onto an existing variant.
2483
- *
2484
- * `failure-mode` is the one exception — its subjects are free-form
2485
- * cluster labels, not loci. The schema preserves them as
2486
- * `{ kind: 'cluster', label }` and the adapters skip them (cluster
2487
- * findings are evidence, not actionable mutations).
2488
- */
2489
-
2490
- /**
2491
- * Discriminated union of every locus the substrate can route findings to.
2492
- *
2493
- * Adapters narrow on `kind` and use the typed components (no string
2494
- * parsing). Adding a variant here REQUIRES updating the parser, the
2495
- * grammar prompt, and at least one adapter — by design.
2496
- */
2497
- type FindingSubject = {
2498
- kind: 'knowledge.wiki';
2499
- slug: string;
2500
- heading?: string;
2501
- } | {
2502
- kind: 'knowledge.claim';
2503
- topic: string;
2504
- } | {
2505
- kind: 'knowledge.raw';
2506
- sourceId: string;
2507
- } | {
2508
- kind: 'knowledge.stale';
2509
- slug: string;
2510
- } | {
2511
- kind: 'system-prompt';
2512
- section: string;
2513
- } | {
2514
- kind: 'skill';
2515
- name: string;
2516
- } | {
2517
- kind: 'tool-doc';
2518
- tool: string;
2519
- aspect?: string;
2520
- } | {
2521
- kind: 'new-tool';
2522
- name: string;
2523
- } | {
2524
- kind: 'mcp';
2525
- server: string;
2526
- tool?: string;
2527
- } | {
2528
- kind: 'hook';
2529
- name: string;
2530
- } | {
2531
- kind: 'subagent';
2532
- name: string;
2533
- } | {
2534
- kind: 'workflow';
2535
- name: string;
2536
- } | {
2537
- kind: 'rollout-policy';
2538
- field: string;
2539
- } | {
2540
- kind: 'agent-profile';
2541
- field: string;
2542
- } | {
2543
- kind: 'code';
2544
- path: string;
2545
- } | {
2546
- kind: 'rag';
2547
- corpus: string;
2548
- docId: string;
2549
- } | {
2550
- kind: 'memory';
2551
- key: string;
2552
- } | {
2553
- kind: 'scaffolding';
2554
- concern: string;
2555
- } | {
2556
- kind: 'output-schema';
2557
- field: string;
2558
- } | {
2559
- kind: 'websearch.outdated';
2560
- topic: string;
2561
- } | {
2562
- kind: 'prior-run-summary';
2563
- topic: string;
2564
- } | {
2565
- kind: 'cluster';
2566
- label: string;
2567
- };
2568
- type FindingSubjectKind = FindingSubject['kind'];
2569
- declare const FINDING_SUBJECT_KINDS: ReadonlyArray<FindingSubjectKind>;
2570
- /**
2571
- * Parse a raw subject string emitted by an analyst kind's actor.
2572
- *
2573
- * Returns the typed `FindingSubject` when `raw` matches the grammar,
2574
- * else `null`. Callers use the `null` return as a signal to either
2575
- * (a) reject the finding at parse time (kinds that emit typed loci —
2576
- * knowledge-gap, improvement, knowledge-poisoning) or (b) lift it as
2577
- * a cluster label (failure-mode).
2578
- *
2579
- * Slugs are constrained to `[a-z0-9-]+` (lowercase kebab) to keep file
2580
- * paths sane downstream. Topics / keys / sections allow any non-empty
2581
- * string (free-form for the LLM's voice) but get trimmed.
2582
- *
2583
- * Empty / whitespace-only inputs return `null`. `undefined` returns
2584
- * `null`. Both are surfaced by the caller as a rejected subject.
2585
- */
2586
- declare function parseFindingSubject(raw: string | null | undefined): FindingSubject | null;
2587
- /**
2588
- * Render the parsed subject back to its canonical string form. Inverse
2589
- * of `parseFindingSubject`; useful when the substrate constructs new
2590
- * findings programmatically (e.g. for tests, replays, or
2591
- * `id_basis` carry-forward).
2592
- */
2593
- declare function renderFindingSubject(s: FindingSubject): string;
2594
- /**
2595
- * The grammar text embedded into kind actor prompts. Kinds opt into
2596
- * the subset of variants they emit (e.g. `improvement` excludes the
2597
- * cluster variant; `failure-mode` includes ONLY the cluster variant).
2598
- *
2599
- * Drift between prompt and parser is impossible: every kind imports
2600
- * this constant + the matching `expects` set, and the unit tests below
2601
- * lock the table to the parser.
2602
- */
2603
- declare const FINDING_SUBJECT_SYNTAX: Readonly<Record<FindingSubjectKind, string>>;
2604
- declare const FINDING_SUBJECT_GRAMMAR_PROMPT: string;
2605
- /**
2606
- * The variants each kind is allowed to emit. Used at the kind factory
2607
- * boundary so a knowledge-gap finding can't sneak in a `system-prompt:*`
2608
- * subject (the improvement-analyst's job) and vice versa.
2609
- *
2610
- * `failure-mode` is restricted to `cluster` — the only kind that emits
2611
- * a non-locus subject.
2612
- */
2613
- declare const KIND_EXPECTED_SUBJECTS: Record<string, ReadonlyArray<FindingSubjectKind>>;
2614
- /** Render only the subject forms one analyst kind is permitted to emit. */
2615
- declare function findingSubjectGrammarPromptFor(kindId: string): string;
2616
- /**
2617
- * Zod schema that validates a raw subject string and returns the parsed
2618
- * `FindingSubject`. Embedded in `RawAnalystFindingSchema` via
2619
- * `transform`, so `subject` arrives at the kind factory either as a
2620
- * typed locus or as a parse error attached to a single Zod issue.
2621
- *
2622
- * Optionality is preserved: subjects ARE optional on the wire (some
2623
- * findings are descriptive, not actionable). When present, they MUST
2624
- * parse — emitting a malformed subject is a contract violation, not a
2625
- * soft signal.
2626
- */
2627
- declare const FindingSubjectStringSchema: z.ZodString;
2628
-
2629
- /**
2630
- * FindingsStore — durable persistence for AnalystFinding rows + a diff
2631
- * helper so we can answer "what changed since the last run?" without
2632
- * recomputing analysts.
2633
- *
2634
- * On-disk shape is JSONL: one finding per line, append-only, locked via
2635
- * LockedJsonlAppender. Operators get crash-safety (no partial JSON),
2636
- * cheap reads (sequential parse), and trivial backup (rsync the file).
2637
- *
2638
- * Reads are non-locking: a reader sees a consistent snapshot of all
2639
- * fully-written lines and skips an incomplete trailing line if the
2640
- * writer is mid-append. Cross-process locking is intentionally out of
2641
- * scope (see locked-jsonl-appender.ts).
2642
- *
2643
- * The store is run-scoped: callers pass `runId` on append and on load,
2644
- * which keeps multi-run files cleanly partitioned. The `diffFindings`
2645
- * helper compares two run-id sets using stable `finding_id` semantics —
2646
- * the diff is the cross-run signal the regression dashboard renders.
2647
- */
2648
-
2649
- /**
2650
- * One persisted row. We attach `run_id` on disk so a single file can
2651
- * hold multiple runs and the diff helper can query without re-walking
2652
- * separate files.
2653
- */
2654
- interface PersistedFinding extends AnalystFinding {
2655
- run_id: string;
2656
- }
2657
- declare class FindingsStore {
2658
- readonly path: string;
2659
- private readonly appender;
2660
- constructor(path: string);
2661
- append(runId: string, findings: AnalystFinding[]): Promise<void>;
2662
- /** Load every persisted finding. Discards malformed trailing lines silently. */
2663
- loadAll(): PersistedFinding[];
2664
- /** Filter to a single run. */
2665
- loadRun(runId: string): PersistedFinding[];
2666
- }
2667
- interface FindingsDiff {
2668
- /** New finding ids in `current` that weren't in `previous`. */
2669
- appeared: PersistedFinding[];
2670
- /** Finding ids in `previous` that aren't in `current`. */
2671
- disappeared: PersistedFinding[];
2672
- /** Same finding id present in both runs and unchanged per the materiality test. */
2673
- persisted: PersistedFinding[];
2674
- /**
2675
- * Same finding id in both runs but at least one non-identity field
2676
- * shifted per `DiffPolicy.isMaterial`. Reported as [previous, current].
2677
- */
2678
- changed: Array<{
2679
- previous: PersistedFinding;
2680
- current: PersistedFinding;
2681
- }>;
2682
- }
2683
- interface DiffPolicy {
2684
- /**
2685
- * Predicate that decides whether two findings (same finding_id) count
2686
- * as a material change. Defaults to {@link defaultIsMaterial}: severity
2687
- * shift, confidence Δ > 0.05, or evidence count change. Compliance /
2688
- * perf consumers MAY supply a stricter predicate (e.g. rationale text
2689
- * diff, metric Δ thresholds).
2690
- */
2691
- isMaterial?: (previous: AnalystFinding, current: AnalystFinding) => boolean;
2692
- }
2693
- /**
2694
- * Default materiality test. Deliberately narrow so LLM-reword churn
2695
- * doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.
2696
- */
2697
- declare function defaultIsMaterial(a: AnalystFinding, b: AnalystFinding): boolean;
2698
- /**
2699
- * Diff two findings sets by stable finding_id. Callers typically load
2700
- * the two run-id slices from the same store and pass them in.
2701
- */
2702
- declare function diffFindings(previous: PersistedFinding[], current: PersistedFinding[], policy?: DiffPolicy): FindingsDiff;
2703
-
2704
- /**
2705
- * Failure-mode analyst — classifies what went wrong and why.
2706
- *
2707
- * Brief: read the trace dataset, identify the top failure modes across
2708
- * runs, classify each with severity + evidence, and surface them as
2709
- * findings. The actor's job is *taxonomy + evidence*, not fix-design —
2710
- * that's the improvement-analyst's job.
2711
- *
2712
- * Eight bounded model subqueries let the actor compare candidate
2713
- * clusters in parallel after it has loaded representative evidence.
2714
- */
2715
-
2716
- declare const FAILURE_MODE_KIND_SPEC: TraceAnalystKindSpec;
2717
-
2718
- /**
2719
- * Improvement analyst — actionable self-improvement findings.
2720
- *
2721
- * Brief: read findings from upstream analysts (failure-mode,
2722
- * knowledge-gap, knowledge-poisoning) AND the trace dataset itself,
2723
- * then propose **concrete edits** to the agent's runtime: prompt
2724
- * additions, RAG documents to ingest, tool descriptions to rewrite,
2725
- * scaffolding changes to make, memory entries to invalidate. Each
2726
- * finding is one proposed edit with the locus, the diff, and the
2727
- * expected effect.
2728
- *
2729
- * This is the self-improvement loop's last mile: the prior
2730
- * kinds describe *what's wrong*; this kind describes *what to change*.
2731
- *
2732
- * Eight bounded model subqueries let the actor compare competing fix
2733
- * directions over the same cited evidence before recommending one.
2734
- */
2735
-
2736
- declare const IMPROVEMENT_KIND_SPEC: TraceAnalystKindSpec;
2737
-
2738
- /**
2739
- * Knowledge-gap analyst — what did the agent NOT know that it needed?
2740
- *
2741
- * Brief: find moments in the trace where the agent had to guess, ask
2742
- * the user to fill in context, recover from a wrong assumption, or
2743
- * loop on a retrieval. Each finding names a *missing or outdated piece
2744
- * of knowledge* the agent's curated knowledge base should have held —
2745
- * or a downstream lookup (web, docs, tool description) that surfaced
2746
- * stale or outdated information.
2747
- *
2748
- * The primary expected store is `@tangle-network/agent-knowledge`: a
2749
- * Karpathy-style wiki the agent maintains with raw ↔ curated pages,
2750
- * source anchors, and claim/relation triples. A gap is anything the
2751
- * agent had to discover at run-time that should already have lived
2752
- * there. Secondary loci: web-search results that returned outdated
2753
- * pages, tool descriptions that omitted critical behavior, system-
2754
- * prompt sections that didn't cover the case.
2755
- *
2756
- * Distinct from failure-mode: failure-mode classifies *how* it broke;
2757
- * knowledge-gap names the *information* whose absence (or staleness)
2758
- * caused the break. One failure-mode often maps to several gaps.
2759
- *
2760
- * Five bounded model subqueries let the actor compare candidate gaps
2761
- * across source layers after it has loaded the relevant excerpts.
2762
- */
2763
-
2764
- declare const KNOWLEDGE_GAP_KIND_SPEC: TraceAnalystKindSpec;
2765
-
2766
- /**
2767
- * Knowledge-poisoning analyst — what FALSE information misled the agent?
2768
- *
2769
- * Brief: find moments where the agent acted on information that was
2770
- * *wrong* — stale memory, RAG documents that contradicted ground truth,
2771
- * tool descriptions that lied about return shapes, system-prompt
2772
- * instructions that no longer matched reality, prior-run summaries that
2773
- * cached a wrong decision.
2774
- *
2775
- * Distinct from knowledge-gap: a gap is "the agent didn't know X"; a
2776
- * poisoning is "the agent confidently used X, but X was wrong." Gaps
2777
- * surface as questions / self-correction; poisonings surface as
2778
- * confident-but-wrong actions that downstream evidence contradicts.
2779
- *
2780
- * Eight bounded model subqueries let the actor independently assess
2781
- * the action and contradiction excerpts for candidate poisonings.
2782
- */
2783
-
2784
- declare const KNOWLEDGE_POISONING_KIND_SPEC: TraceAnalystKindSpec;
2785
-
2786
- /**
2787
- * Default analyst kinds focused on agent failure + recursive
2788
- * self-improvement.
2789
- *
2790
- * The four kinds chain: failure-mode classifies; knowledge-gap and
2791
- * knowledge-poisoning explain *why* in two orthogonal ways; improvement
2792
- * proposes concrete edits. Register all four against the same trace
2793
- * store in this order and run the registry with `chainFindings: true`
2794
- * to pass each completed kind's findings to the kinds that follow it.
2795
- */
2796
-
2797
- /**
2798
- * The default kind suite. Order is the run order operators should
2799
- * use: failure-mode first (no upstream deps), gap + poisoning next
2800
- * (both depend on failures), improvement last (chains all three).
2801
- */
2802
- declare const DEFAULT_TRACE_ANALYST_KINDS: readonly TraceAnalystKindSpec[];
2803
-
2804
- /**
2805
- * Skill-usage analyst — a DETERMINISTIC `Analyst` over a Claude/Codex skill
2806
- * library + its trace corpus. Unlike the trace-store kinds (failure-mode,
2807
- * improvement, ...) this kind calls no LLM: it mines real usage and skill
2808
- * structure and emits findings by rule.
2809
- *
2810
- * It exists because the naive "Skill-tool invocation count" lies low — it
2811
- * misses orchestrated sub-dispatch (a leaf skill run BY /pursue or /governor
2812
- * logs under the parent), slash-command entry, local-script bypass, and
2813
- * on-disk artifacts. The 2026-05-30 skill audit found 39/53 skills at zero
2814
- * direct invocations, yet only one was a genuine cut: the rest were
2815
- * measurement-invisible or discovery-limited. This analyst encodes that
2816
- * lesson as a multi-signal usage model so a cheap repeatable pass can keep
2817
- * the library honest, and so the expensive audit workflow's verdicts can
2818
- * GEPA-distill it toward agreement (see `gold/skill-verdicts.gold.jsonl`).
2819
- *
2820
- * Report-building (`buildSkillUsageReport`, an fs scan) is separated from
2821
- * finding emission (`SkillUsageAnalyst.analyze`, pure) so the slow scan runs
2822
- * once at the registry boundary and the rule logic stays unit-testable.
2823
- */
2824
-
2825
- type SkillKind = 'public' | 'private';
2826
- /** One skill's multi-signal usage + structure. All counts are deterministic. */
2827
- interface SkillUsageRecord {
2828
- name: string;
2829
- kind: SkillKind;
2830
- /** Absolute path to the skill's SKILL.md. */
2831
- path: string;
2832
- lines: number;
2833
- /** `"skill":"<name>"` Skill-tool invocations across the trace corpus. */
2834
- directInvocations: number;
2835
- /** `<command-name>/<name>` slash invocations across the trace corpus. */
2836
- slashInvocations: number;
2837
- /** Sibling skills whose SKILL.md dispatches to this one (`/<name>`). Proxy
2838
- * for orchestrated sub-dispatch the per-skill counter cannot see. */
2839
- inboundRefs: number;
2840
- /** On-disk artifacts attributable to the skill (e.g. `.evolve/<name>/**`). */
2841
- artifactCount: number;
2842
- /** Tangle-private reference count in the body (leak signal for public skills). */
2843
- tanglePrivateRefs: number;
2844
- hasReferencesDir: boolean;
2845
- hasEvalsDir: boolean;
2846
- /** Body mentions `skill-runs.jsonl` (visible to /reflect + /governor). */
2847
- logsRuns: boolean;
2848
- /** Description carries an explicit `Triggers:` clause / trigger phrases. */
2849
- hasTriggerPhrases: boolean;
2850
- }
2851
- interface SkillUsageReport {
2852
- generatedFromTraces: number;
2853
- records: SkillUsageRecord[];
2854
- }
2855
- interface SkillUsageScanConfig {
2856
- /** Dirs holding `*.jsonl` transcripts (Claude `~/.claude/projects`, Codex sessions). */
2857
- transcriptDirs: string[];
2858
- /** Skill roots to scan; each dir directly under `root` with a `SKILL.md` is a skill. */
2859
- skillRoots: {
2860
- root: string;
2861
- kind: SkillKind;
2862
- }[];
2863
- /** Roots scanned for `<root>/.evolve/<skill>` artifact dirs. */
2864
- artifactRoots?: string[];
2865
- /** Token-prefixed mappings: skill name → extra artifact subpaths under an artifactRoot
2866
- * (e.g. reflect → `.evolve/reflections`). Catches non-eponymous artifact dirs. */
2867
- artifactAliases?: Record<string, string[]>;
2868
- /** Cap files read per transcript dir (bounds a huge corpus); 0 = unbounded. */
2869
- maxTranscriptsPerDir?: number;
2870
- }
2871
- /** Scan the corpus + skill roots into a {@link SkillUsageReport}. Deterministic. */
2872
- declare function buildSkillUsageReport(config: SkillUsageScanConfig): SkillUsageReport;
2873
- /** Pure rule pass over a report → findings. Exported for direct/unit use. */
2874
- declare function emitSkillUsageFindings(report: SkillUsageReport, producedAt: string): AnalystFinding[];
2875
- declare class SkillUsageAnalyst implements Analyst<SkillUsageReport> {
2876
- readonly id = "skill-usage";
2877
- readonly description = "Deterministic multi-signal skill-usage analysis: flags dead skills, measurement-invisible (orchestrated) usage, discovery gaps, public-repo leaks, bloat, missing evals, and missing run-logging.";
2878
- readonly inputKind: "custom";
2879
- readonly cost: {
2880
- kind: "deterministic";
2881
- est_usd_per_run: number;
2882
- };
2883
- readonly version = "1.0.0";
2884
- analyze(input: SkillUsageReport, ctx: AnalystContext): Promise<AnalystFinding[]>;
2885
- }
2886
- declare const SKILL_USAGE_ANALYST: SkillUsageAnalyst;
2887
-
63
+ //#endregion
64
+ //#region src/analyst/parse-tolerant.d.ts
2888
65
  /**
2889
66
  * Forgiving pre-parse for analyst findings. Weak models routinely emit
2890
67
  * schema-correct content in an unusable wrapper — fenced ```json blocks, a
@@ -2907,10 +84,11 @@ declare function coerceJson(text: string): unknown;
2907
84
  * Coerce arbitrary actor/structurer output into an array of candidate finding
2908
85
  * rows: a JSON string → parse; a single object → 1-element array; an array →
2909
86
  * as-is; anything else → []. Callers still run each row through Zod
2910
- * (`parseCanonicalRawFinding`) — this only fixes the SHAPE, never invents fields.
87
+ * (`parseRawFinding`) — this only fixes the shape and never invents fields.
2911
88
  */
2912
89
  declare function coerceToFindingRows(raw: unknown): unknown[];
2913
-
90
+ //#endregion
91
+ //#region src/analyst/steer-firewall.d.ts
2914
92
  /** DESCRIPTIVE predicate: does the finding cite at least one observable
2915
93
  * (span/event/artifact) evidence ref. Useful for ranking evidence quality or
2916
94
  * rendering — it is NOT the steer gate. Evidence presence is the WRONG
@@ -2944,79 +122,51 @@ declare function isJudgeVerdict(finding: AnalystFinding): boolean;
2944
122
  * compile-time tripwire on the obvious direct channel.
2945
123
  */
2946
124
  declare function assertNoJudgeVerdict(findings: ReadonlyArray<AnalystFinding>, context?: string): ReadonlyArray<AnalystFinding>;
2947
-
2948
- /**
2949
- * `structureFindings` — the deferred structuring pass (DSPy TwoStepAdapter /
2950
- * HALO `synthesize_traces` analog). The agentic actor reasons FREE-FORM and
2951
- * emits a prose `report` (which any model does reliably); this separate, cheap
2952
- * call's ONLY job is to turn that report into `AnalystFinding[]`. Decoupling
2953
- * reasoning from structuring is what makes the SEMANTIC findings model-agnostic
2954
- * — the reasoning model never has to satisfy a strict typed-array contract
2955
- * while it diagnoses.
2956
- *
2957
- * Forgiving: the response runs through `coerceToFindingRows` (de-fence, lift
2958
- * single→array) before Zod, and on a zero-finding extraction from a substantive
2959
- * report it reasks ONCE with the schema restated. Returns a typed outcome so a
2960
- * legitimate "nothing to report" is distinguishable from a failed extraction
2961
- * (no silent empty).
2962
- */
2963
-
125
+ //#endregion
126
+ //#region src/analyst/structure-findings.d.ts
2964
127
  interface StructureFindingsOptions {
2965
- /** The actor's free-form diagnosis prose. */
2966
- report: string;
2967
- analystId: string;
2968
- /** Coarse classification stamped on every extracted finding. */
2969
- area: string;
2970
- model: string;
2971
- baseUrl: string;
2972
- apiKey?: string;
2973
- /** Optional ledger for direct use. */
2974
- costLedger?: CostLedgerHandle;
2975
- costPhase?: string;
2976
- costTags?: Record<string, string>;
2977
- maxTokens?: number;
2978
- signal?: AbortSignal;
2979
- /** Max reask attempts after a zero/invalid extraction. Default 1. */
2980
- maxReasks?: number;
2981
- /** Apply the caller's normal finding rules before a recovered row is lifted. */
2982
- processRow?: (row: RawAnalystFinding) => RawAnalystFinding | null;
2983
- /** Apply canonical multi-citation rules after any original callback. */
2984
- processCanonicalRow?: (row: CanonicalRawAnalystFinding) => CanonicalRawAnalystFinding | null;
2985
- /** Provenance copied onto every recovered finding. */
2986
- findingMetadata?: Record<string, unknown>;
2987
- /** Test seam: inject a fetch (no network in unit tests). */
2988
- fetchImpl?: LlmClientOptions['fetch'];
128
+ /** The actor's free-form diagnosis prose. */
129
+ report: string;
130
+ analystId: string;
131
+ /** Coarse classification stamped on every extracted finding. */
132
+ area: string;
133
+ model: string;
134
+ baseUrl: string;
135
+ apiKey?: string;
136
+ /** Optional ledger for direct use. */
137
+ costLedger?: CostLedgerHandle;
138
+ costPhase?: string;
139
+ costTags?: Record<string, string>;
140
+ maxTokens?: number;
141
+ signal?: AbortSignal;
142
+ /** Max reask attempts after a zero/invalid extraction. Default 1. */
143
+ maxReasks?: number;
144
+ /** Apply the caller's normal finding rules before a recovered row is lifted. */
145
+ processRow?: (row: RawAnalystFinding) => RawAnalystFinding | null;
146
+ /** Provenance copied onto every recovered finding. */
147
+ findingMetadata?: Record<string, unknown>;
148
+ /** Test seam: inject a fetch (no network in unit tests). */
149
+ fetchImpl?: LlmClientOptions['fetch'];
2989
150
  }
2990
151
  interface StructureFindingsResult {
2991
- findings: AnalystFinding[];
2992
- outcome: 'ok' | 'extraction_failed';
152
+ findings: AnalystFinding[];
153
+ outcome: 'ok' | 'extraction_failed';
2993
154
  }
2994
155
  declare function structureFindings(opts: StructureFindingsOptions): Promise<StructureFindingsResult>;
2995
-
2996
- /**
2997
- * Pre-curated tool subsets for analyst kinds.
2998
- *
2999
- * The full trace-analyst tool set is seven functions. Most kinds only
3000
- * need three or four. Picking from named groups instead of importing
3001
- * the whole bundle keeps every kind's actor-context budget tight and
3002
- * makes "what can this analyst see?" obvious at registration time.
3003
- *
3004
- * Each function in the group keeps its full `name`/`description` from
3005
- * `buildTraceAnalystTools` — we filter, we don't re-implement.
3006
- */
3007
-
156
+ //#endregion
157
+ //#region src/analyst/tool-groups.d.ts
3008
158
  /** Named tool sets. Kinds pass `tools: TRACE_TOOL_GROUPS.failureForensics` etc. */
3009
- type TraceToolGroupName =
159
+ type TraceToolGroupName =
3010
160
  /** All seven tools. Use for open-ended discovery kinds. */
3011
- 'all'
161
+ 'all' |
3012
162
  /** Overview + paginated query + count. No deep reads. Cheap. */
3013
- | 'discovery'
163
+ 'discovery' |
3014
164
  /** Discovery + viewTrace + viewSpans. Deep-read but no regex search. */
3015
- | 'discoveryAndRead'
165
+ 'discoveryAndRead' |
3016
166
  /** Discovery + search tools. For pattern-matching across many traces. */
3017
- | 'discoveryAndSearch'
167
+ 'discoveryAndSearch' |
3018
168
  /** Discovery + viewSpans + searchSpan. Targeted-span work after another kind narrows down. */
3019
- | 'targeted';
169
+ 'targeted';
3020
170
  /**
3021
171
  * Build the tool set for a named group bound to a specific trace store.
3022
172
  *
@@ -3025,5 +175,6 @@ type TraceToolGroupName =
3025
175
  * silently returning all tools would defeat the cost-control point.
3026
176
  */
3027
177
  declare function buildTraceToolsForGroup(group: TraceToolGroupName, store: TraceAnalysisStore): AxFunction[];
3028
-
3029
- export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type CanonicalRawAnalystFinding, CanonicalRawAnalystFindingSchema, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type MockTransportOpts, type PersistedFinding, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystGolden, type TraceAnalystKindSpec, type TraceToolGroupName, type VerifierAdapterOpts, assertNoJudgeVerdict, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isJudgeVerdict, isTraceObservable, liftSeverity, makeFinding, parseCanonicalRawFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
178
+ //#endregion
179
+ export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, type CustomTransportOpts, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type MockTransportOpts, type PersistedFinding, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystGolden, type TraceAnalystKindSpec, type TraceToolGroupName, type VerifierAdapterOpts, assertNoJudgeVerdict, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isJudgeVerdict, isTraceObservable, liftSeverity, makeFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
180
+ //# sourceMappingURL=index.d.ts.map