@tangle-network/agent-eval 0.136.0 → 0.138.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (231) hide show
  1. package/CHANGELOG.md +86 -1
  2. package/README.md +37 -2
  3. package/dist/{agent-profile-cell-OhuTee9n.js → agent-profile-cell-CbfBm2g6.js} +2 -2
  4. package/dist/{agent-profile-cell-OhuTee9n.js.map → agent-profile-cell-CbfBm2g6.js.map} +1 -1
  5. package/dist/{agent-profile-cell-CCm3l2v2.d.ts → agent-profile-cell-Cw0PVwDr.d.ts} +2 -2
  6. package/dist/{agent-profile-cell-CCm3l2v2.d.ts.map → agent-profile-cell-Cw0PVwDr.d.ts.map} +1 -1
  7. package/dist/analyst/index.d.ts +574 -18
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +25 -5
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{analyze-runs-jjCmF8pU.js → analyze-runs-BScZvqMV.js} +6 -6
  12. package/dist/{analyze-runs-jjCmF8pU.js.map → analyze-runs-BScZvqMV.js.map} +1 -1
  13. package/dist/{analyze-runs-Cda5Xkj1.d.ts → analyze-runs-CPYxfPWT.d.ts} +6 -6
  14. package/dist/{analyze-runs-Cda5Xkj1.d.ts.map → analyze-runs-CPYxfPWT.d.ts.map} +1 -1
  15. package/dist/{baseline-BUeFcgrn.js → baseline-C-GocmIW.js} +2 -2
  16. package/dist/{baseline-BUeFcgrn.js.map → baseline-C-GocmIW.js.map} +1 -1
  17. package/dist/benchmark-D8dkki-J.js +554 -0
  18. package/dist/benchmark-D8dkki-J.js.map +1 -0
  19. package/dist/benchmark-DlQgU_XI.d.ts +236 -0
  20. package/dist/benchmark-DlQgU_XI.d.ts.map +1 -0
  21. package/dist/benchmark-command-CMqVqReF.js +4332 -0
  22. package/dist/benchmark-command-CMqVqReF.js.map +1 -0
  23. package/dist/benchmarks/index.d.ts +1 -1
  24. package/dist/benchmarks/index.js +1 -1
  25. package/dist/{benchmarks-Dfgm9ts5.js → benchmarks-BJ_xK5rQ.js} +4 -3
  26. package/dist/{benchmarks-Dfgm9ts5.js.map → benchmarks-BJ_xK5rQ.js.map} +1 -1
  27. package/dist/builder-eval/index.js +2 -2
  28. package/dist/campaign/index.d.ts +6 -6
  29. package/dist/campaign/index.js +4 -4
  30. package/dist/{campaign-Dz8uQnhC.js → campaign-BIBS-NHV.js} +219 -78
  31. package/dist/campaign-BIBS-NHV.js.map +1 -0
  32. package/dist/cli.js +9 -2
  33. package/dist/cli.js.map +1 -1
  34. package/dist/client-BwPKohkJ.d.ts +202 -0
  35. package/dist/client-BwPKohkJ.d.ts.map +1 -0
  36. package/dist/completion-verifier-B4-IMYcS.d.ts +240 -0
  37. package/dist/completion-verifier-B4-IMYcS.d.ts.map +1 -0
  38. package/dist/contract/index.d.ts +10 -9
  39. package/dist/contract/index.d.ts.map +1 -1
  40. package/dist/contract/index.js +13 -13
  41. package/dist/control.d.ts +2 -2
  42. package/dist/control.js +1 -1
  43. package/dist/{cost-ledger-fGS_u_O1.d.ts → cost-ledger-B1D3COAc.d.ts} +6 -5
  44. package/dist/{cost-ledger-fGS_u_O1.d.ts.map → cost-ledger-B1D3COAc.d.ts.map} +1 -1
  45. package/dist/{cost-ledger-DHAjwNj7.js → cost-ledger-CHDLA0Ss.js} +91 -46
  46. package/dist/cost-ledger-CHDLA0Ss.js.map +1 -0
  47. package/dist/{dataset-BvtnC8Dc.d.ts → dataset-v_Y5902-.d.ts} +2 -2
  48. package/dist/{dataset-BvtnC8Dc.d.ts.map → dataset-v_Y5902-.d.ts.map} +1 -1
  49. package/dist/{default-registry-Brxr728w.d.ts → default-registry-PUhIVRWz.d.ts} +77 -138
  50. package/dist/default-registry-PUhIVRWz.d.ts.map +1 -0
  51. package/dist/{default-registry-CHmdy2An.js → default-registry-lp5R0lve.js} +1617 -291
  52. package/dist/default-registry-lp5R0lve.js.map +1 -0
  53. package/dist/{errors-8YnH8WlF.js → errors-D-LKuDhb.js} +8 -2
  54. package/dist/errors-D-LKuDhb.js.map +1 -0
  55. package/dist/{errors-CEk209JS.d.ts → errors-DkfjIDvD.d.ts} +9 -3
  56. package/dist/errors-DkfjIDvD.d.ts.map +1 -0
  57. package/dist/{eval-campaign-Cc8WZJ6b.js → eval-campaign-9MozgKL7.js} +6 -6
  58. package/dist/{eval-campaign-Cc8WZJ6b.js.map → eval-campaign-9MozgKL7.js.map} +1 -1
  59. package/dist/exact-types-Dpw2LeHA.d.ts +234 -0
  60. package/dist/exact-types-Dpw2LeHA.d.ts.map +1 -0
  61. package/dist/{extract-usage-DIQpN-ww.js → extract-usage-CS391dOE.js} +3 -3
  62. package/dist/{extract-usage-DIQpN-ww.js.map → extract-usage-CS391dOE.js.map} +1 -1
  63. package/dist/{feedback-trajectory-CVaeREXV.d.ts → feedback-trajectory-CoNep7rl.d.ts} +91 -3
  64. package/dist/feedback-trajectory-CoNep7rl.d.ts.map +1 -0
  65. package/dist/fuzz.d.ts +1 -1
  66. package/dist/fuzz.js +2 -2
  67. package/dist/hosted/index.d.ts +3 -2
  68. package/dist/hosted/index.d.ts.map +1 -1
  69. package/dist/{index-AbhwHp0V.d.ts → index-B2-IxCMB.d.ts} +2 -2
  70. package/dist/{index-AbhwHp0V.d.ts.map → index-B2-IxCMB.d.ts.map} +1 -1
  71. package/dist/index-BipJlj-C.d.ts +316 -0
  72. package/dist/index-BipJlj-C.d.ts.map +1 -0
  73. package/dist/{index-CQsJcqch.d.ts → index-CjVYlVBK.d.ts} +5 -5
  74. package/dist/{index-CQsJcqch.d.ts.map → index-CjVYlVBK.d.ts.map} +1 -1
  75. package/dist/{index-B4Fjfo5U.d.ts → index-D0cxAdaV.d.ts} +89 -317
  76. package/dist/index-D0cxAdaV.d.ts.map +1 -0
  77. package/dist/{index-DuhJaaiH.d.ts → index-DEb46kc6.d.ts} +2 -2
  78. package/dist/index-DEb46kc6.d.ts.map +1 -0
  79. package/dist/{index-C2fkZhv_.d.ts → index-sMN_hI4E.d.ts} +3 -3
  80. package/dist/{index-C2fkZhv_.d.ts.map → index-sMN_hI4E.d.ts.map} +1 -1
  81. package/dist/index.d.ts +30 -70
  82. package/dist/index.d.ts.map +1 -1
  83. package/dist/index.js +175 -35
  84. package/dist/index.js.map +1 -1
  85. package/dist/{client-DcvgkaZi.d.ts → insight-report-CXd8VBDR.d.ts} +5 -203
  86. package/dist/insight-report-CXd8VBDR.d.ts.map +1 -0
  87. package/dist/{integrity-rmVhXWA7.d.ts → integrity-B-MLFz0I.d.ts} +3 -3
  88. package/dist/{integrity-rmVhXWA7.d.ts.map → integrity-B-MLFz0I.d.ts.map} +1 -1
  89. package/dist/integrity-CCXTftiL.js +1360 -0
  90. package/dist/integrity-CCXTftiL.js.map +1 -0
  91. package/dist/{integrity-BzRbCHzi.js → integrity-fdt8XPAv.js} +2 -2
  92. package/dist/{integrity-BzRbCHzi.js.map → integrity-fdt8XPAv.js.map} +1 -1
  93. package/dist/ledger-core/index.d.ts +1 -1
  94. package/dist/ledger-core/index.js +1 -1
  95. package/dist/{ledger-core-DAKFKRzi.js → ledger-core-C0Yx1I14.js} +303 -110
  96. package/dist/ledger-core-C0Yx1I14.js.map +1 -0
  97. package/dist/{llm-client-DHx8pzyJ.js → llm-client-Cj3c7PEm.js} +6 -6
  98. package/dist/llm-client-Cj3c7PEm.js.map +1 -0
  99. package/dist/meta-eval/index.d.ts +2 -2
  100. package/dist/meta-eval/index.js +3 -3
  101. package/dist/{mint-DyRUc9k6.js → mint-Ctwk079K.js} +4 -4
  102. package/dist/{mint-DyRUc9k6.js.map → mint-Ctwk079K.js.map} +1 -1
  103. package/dist/multishot/index.d.ts +2 -2
  104. package/dist/openapi.json +1 -1
  105. package/dist/{paired-arms-BbFKrAU-.js → paired-arms-iZ08VFMN.js} +3 -3
  106. package/dist/{paired-arms-BbFKrAU-.js.map → paired-arms-iZ08VFMN.js.map} +1 -1
  107. package/dist/pipelines/index.js +2 -2
  108. package/dist/profile-cell.d.ts +1 -1
  109. package/dist/profile-cell.js +1 -1
  110. package/dist/{proposal-findings-DCawte-y.js → proposal-findings-2GIUo1et.js} +2 -68
  111. package/dist/proposal-findings-2GIUo1et.js.map +1 -0
  112. package/dist/{propose-review-control-SQ-n9-We.js → propose-review-control-DLXz4FCX.js} +2 -2
  113. package/dist/{propose-review-control-SQ-n9-We.js.map → propose-review-control-DLXz4FCX.js.map} +1 -1
  114. package/dist/registry-C4yJTza7.d.ts +178 -0
  115. package/dist/registry-C4yJTza7.d.ts.map +1 -0
  116. package/dist/{release-report-DooPguBc.js → release-report-B5XPBvAU.js} +4 -4
  117. package/dist/{release-report-DooPguBc.js.map → release-report-B5XPBvAU.js.map} +1 -1
  118. package/dist/{release-report-DpBxGGI1.d.ts → release-report-CoyvyLBs.d.ts} +4 -4
  119. package/dist/{release-report-DpBxGGI1.d.ts.map → release-report-CoyvyLBs.d.ts.map} +1 -1
  120. package/dist/{replay-C6wRg47C.js → replay-Cb-4Vf0k.js} +249 -8
  121. package/dist/replay-Cb-4Vf0k.js.map +1 -0
  122. package/dist/{replay-BRfMIs81.d.ts → replay-DbIYwso6.d.ts} +227 -52
  123. package/dist/replay-DbIYwso6.d.ts.map +1 -0
  124. package/dist/reporting.d.ts +4 -4
  125. package/dist/reporting.js +4 -4
  126. package/dist/{researcher-Doo95b50.d.ts → researcher-BCeOEjtR.d.ts} +6 -7
  127. package/dist/researcher-BCeOEjtR.d.ts.map +1 -0
  128. package/dist/{reward-hacking-a-kYs0-i.js → reward-hacking-GyN0kMd8.js} +3 -3
  129. package/dist/{reward-hacking-a-kYs0-i.js.map → reward-hacking-GyN0kMd8.js.map} +1 -1
  130. package/dist/{reward-hacking-D-QqXvg-.d.ts → reward-hacking-sE2l_NV6.d.ts} +2 -2
  131. package/dist/{reward-hacking-D-QqXvg-.d.ts.map → reward-hacking-sE2l_NV6.d.ts.map} +1 -1
  132. package/dist/rl.d.ts +6 -6
  133. package/dist/rl.js +9 -9
  134. package/dist/rollout/index.d.ts +1 -1
  135. package/dist/rollout/index.js +3 -3
  136. package/dist/{rollout-DLSUIWLu.js → rollout-DQFl0UXA.js} +2 -2
  137. package/dist/{rollout-DLSUIWLu.js.map → rollout-DQFl0UXA.js.map} +1 -1
  138. package/dist/{rubric-predictive-validity-BJf-8ejY.js → rubric-predictive-validity-BRR632r1.js} +2 -2
  139. package/dist/{rubric-predictive-validity-BJf-8ejY.js.map → rubric-predictive-validity-BRR632r1.js.map} +1 -1
  140. package/dist/{rubric-predictive-validity-C1dCLcvb.d.ts → rubric-predictive-validity-w2klGv1u.d.ts} +2 -2
  141. package/dist/{rubric-predictive-validity-C1dCLcvb.d.ts.map → rubric-predictive-validity-w2klGv1u.d.ts.map} +1 -1
  142. package/dist/{run-evidence-DokQtX0-.d.ts → run-evidence-CbE0A8Xg.d.ts} +3 -3
  143. package/dist/{run-evidence-DokQtX0-.d.ts.map → run-evidence-CbE0A8Xg.d.ts.map} +1 -1
  144. package/dist/{run-record-DcObtIGh.d.ts → run-record-DwHMk1Ai.d.ts} +4 -4
  145. package/dist/{run-record-DcObtIGh.d.ts.map → run-record-DwHMk1Ai.d.ts.map} +1 -1
  146. package/dist/{run-record-BIwU2wdV.js → run-record-vRgqWmJw.js} +3 -3
  147. package/dist/{run-record-BIwU2wdV.js.map → run-record-vRgqWmJw.js.map} +1 -1
  148. package/dist/{semantic-concept-judge-Btozx3Vc.js → semantic-concept-judge-DYXDPZW0.js} +12 -6
  149. package/dist/semantic-concept-judge-DYXDPZW0.js.map +1 -0
  150. package/dist/{server-Bz3WQJs6.js → server-DLEvyW2z.js} +3 -3
  151. package/dist/{server-Bz3WQJs6.js.map → server-DLEvyW2z.js.map} +1 -1
  152. package/dist/single-run-lock-D_bS5xhj.js +318 -0
  153. package/dist/single-run-lock-D_bS5xhj.js.map +1 -0
  154. package/dist/{skill-usage-BDQVPIG1.d.ts → skill-usage-Bv3G4VkA.d.ts} +36 -48
  155. package/dist/skill-usage-Bv3G4VkA.d.ts.map +1 -0
  156. package/dist/{skillopt-optimization-method-0UmPD6aP.js → skillopt-optimization-method-CjKMZy0d.js} +10 -185
  157. package/dist/skillopt-optimization-method-CjKMZy0d.js.map +1 -0
  158. package/dist/{skillopt-optimization-method-CwSYkv35.d.ts → skillopt-optimization-method-CzfnA8O-.d.ts} +11 -12
  159. package/dist/skillopt-optimization-method-CzfnA8O-.d.ts.map +1 -0
  160. package/dist/{statistics-CnGCLLqc.js → statistics-ByxzSiOM.js} +2 -2
  161. package/dist/{statistics-CnGCLLqc.js.map → statistics-ByxzSiOM.js.map} +1 -1
  162. package/dist/{statistics-CKOqre5S.d.ts → statistics-mf70aXKp.d.ts} +2 -2
  163. package/dist/{statistics-CKOqre5S.d.ts.map → statistics-mf70aXKp.d.ts.map} +1 -1
  164. package/dist/store-otlp-BenKynPE.js +1688 -0
  165. package/dist/store-otlp-BenKynPE.js.map +1 -0
  166. package/dist/{summary-report-BEk8OFLs.js → summary-report-9A5y7EsK.js} +4 -4
  167. package/dist/{summary-report-BEk8OFLs.js.map → summary-report-9A5y7EsK.js.map} +1 -1
  168. package/dist/{summary-report-CPMINBqs.d.ts → summary-report-BKinV4yD.d.ts} +3 -3
  169. package/dist/{summary-report-CPMINBqs.d.ts.map → summary-report-BKinV4yD.d.ts.map} +1 -1
  170. package/dist/supervisor-run/index.d.ts +3 -2
  171. package/dist/supervisor-run/index.js +3 -2
  172. package/dist/{supervisor-run-Dr5HnTup.js → supervisor-run-B2EWUmQY.js} +28 -454
  173. package/dist/supervisor-run-B2EWUmQY.js.map +1 -0
  174. package/dist/{test-graded-scenario-BsqWLmPt.js → test-graded-scenario-JHcKQNpq.js} +2 -2
  175. package/dist/{test-graded-scenario-BsqWLmPt.js.map → test-graded-scenario-JHcKQNpq.js.map} +1 -1
  176. package/dist/tools-DZGdROtG.js +255 -0
  177. package/dist/tools-DZGdROtG.js.map +1 -0
  178. package/dist/traces.d.ts +6 -7
  179. package/dist/traces.js +6 -6
  180. package/dist/types-5q2T25iW.d.ts +804 -0
  181. package/dist/types-5q2T25iW.d.ts.map +1 -0
  182. package/dist/{types-DVjczBM9.d.ts → types-BtJhn8v6.d.ts} +260 -6
  183. package/dist/types-BtJhn8v6.d.ts.map +1 -0
  184. package/dist/{index-CyC1BTmn.d.ts → types-Dea6tiVI.d.ts} +16 -238
  185. package/dist/types-Dea6tiVI.d.ts.map +1 -0
  186. package/dist/{types-DiWLru6Z.d.ts → types-zFYez3PK.d.ts} +5 -5
  187. package/dist/{types-DiWLru6Z.d.ts.map → types-zFYez3PK.d.ts.map} +1 -1
  188. package/dist/wire/index.d.ts +3 -3
  189. package/dist/wire/index.js +1 -1
  190. package/docs/feedback-trajectories.md +100 -1
  191. package/docs/trace-analysis.md +494 -58
  192. package/package.json +9 -3
  193. package/dist/analyst-BkTS3C58.d.ts +0 -89
  194. package/dist/analyst-BkTS3C58.d.ts.map +0 -1
  195. package/dist/analyst-j5je5J7c.js +0 -152
  196. package/dist/analyst-j5je5J7c.js.map +0 -1
  197. package/dist/campaign-Dz8uQnhC.js.map +0 -1
  198. package/dist/client-DcvgkaZi.d.ts.map +0 -1
  199. package/dist/concurrency-MUjT7VjM.js +0 -109
  200. package/dist/concurrency-MUjT7VjM.js.map +0 -1
  201. package/dist/cost-ledger-DHAjwNj7.js.map +0 -1
  202. package/dist/default-registry-Brxr728w.d.ts.map +0 -1
  203. package/dist/default-registry-CHmdy2An.js.map +0 -1
  204. package/dist/errors-8YnH8WlF.js.map +0 -1
  205. package/dist/errors-CEk209JS.d.ts.map +0 -1
  206. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +0 -1
  207. package/dist/index-B4Fjfo5U.d.ts.map +0 -1
  208. package/dist/index-CyC1BTmn.d.ts.map +0 -1
  209. package/dist/index-DuhJaaiH.d.ts.map +0 -1
  210. package/dist/ledger-core-DAKFKRzi.js.map +0 -1
  211. package/dist/llm-client-BiK4HW0u.d.ts +0 -290
  212. package/dist/llm-client-BiK4HW0u.d.ts.map +0 -1
  213. package/dist/llm-client-DHx8pzyJ.js.map +0 -1
  214. package/dist/proposal-findings-DCawte-y.js.map +0 -1
  215. package/dist/raw-provider-sink-BU29Sh8h.d.ts +0 -134
  216. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +0 -1
  217. package/dist/replay-BRfMIs81.d.ts.map +0 -1
  218. package/dist/replay-C6wRg47C.js.map +0 -1
  219. package/dist/researcher-Doo95b50.d.ts.map +0 -1
  220. package/dist/semantic-concept-judge-Btozx3Vc.js.map +0 -1
  221. package/dist/skill-usage-BDQVPIG1.d.ts.map +0 -1
  222. package/dist/skillopt-optimization-method-0UmPD6aP.js.map +0 -1
  223. package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +0 -1
  224. package/dist/store-CxJry_cs.d.ts +0 -229
  225. package/dist/store-CxJry_cs.d.ts.map +0 -1
  226. package/dist/supervisor-run-Dr5HnTup.js.map +0 -1
  227. package/dist/tools-D8yTtNSN.js +0 -1190
  228. package/dist/tools-D8yTtNSN.js.map +0 -1
  229. package/dist/types-Cc3qbqzj.d.ts +0 -387
  230. package/dist/types-Cc3qbqzj.d.ts.map +0 -1
  231. package/dist/types-DVjczBM9.d.ts.map +0 -1
@@ -1,52 +1,81 @@
1
1
  # Trace Analysis
2
2
 
3
- Trace analysis is the bridge between raw product telemetry and useful eval work.
3
+ Trace analysis answers three different questions:
4
4
 
5
- ```txt
6
- live product run
7
- -> TraceEmitter / TraceStore
8
- -> TraceAnalyst investigates trace corpora
9
- -> findings become ASI, failures, replay cases, and release actions
10
- ```
5
+ 1. What happened in the run?
6
+ 2. Which exact steps support a suspected problem?
7
+ 3. Does an analyst find labeled problems reliably enough to use?
8
+
9
+ Keep those answers separate.
10
+ A generated finding is a review request, not training truth.
11
+
12
+ ## Run The Built-In Analysts
13
+
14
+ The default registry always includes deterministic checks.
15
+ Model-assisted analysts are added only when you provide a model client.
11
16
 
12
- ## When To Use TraceAnalyst
17
+ ```ts
18
+ import {
19
+ buildDefaultAnalystRegistry,
20
+ } from '@tangle-network/agent-eval/analyst'
21
+ import { OtlpFileTraceStore } from '@tangle-network/agent-eval/traces'
13
22
 
14
- Use `TraceAnalyst` when you have more than a few traces and need to answer:
23
+ const traceStore = new OtlpFileTraceStore({ path: 'traces.otlp.jsonl' })
24
+ const analysts = buildDefaultAnalystRegistry()
15
25
 
16
- - which failure modes are recurring?
17
- - which spans explain a regression?
18
- - did retrieval, integrations, sandbox, or policy block the run?
19
- - are failed runs missing evidence that the optimizer needs?
20
- - which product surfaces deserve the next fix?
26
+ const result = await analysts.run('release-42', { traceStore })
21
27
 
22
- Use summary tables and release confidence for promotion decisions. Use
23
- TraceAnalyst to explain the evidence behind those decisions.
28
+ for (const finding of result.findings) {
29
+ console.log(finding.claim, finding.evidence_refs)
30
+ }
31
+ ```
24
32
 
25
- ## Minimal Flow
33
+ Use `result.per_analyst` to inspect failures, latency, calls, tokens, and cost.
34
+ An analyst failure is recorded separately from an agent failure.
35
+
36
+ Products can implement `TraceAnalysisStore`; they do not need to use the file store in production.
37
+ Custom stores provide `hasTrace` and batched `hasSpans` alongside the seven reads, and accept a `TraceAnalysisStoreContext` so cancellation reaches storage and scans.
38
+ The binding validates every custom-store result.
39
+ Missing fields, undeclared fields, inconsistent counts, and false continuation flags throw `TraceAnalysisStoreContractError` with code `backend_integrity`.
40
+ The analyst runs one Ax executor loop and accepts only an explicit structured `final(task, { report, findings })` result; max-turn fallback text fails loud.
41
+
42
+ ### Bind the same reads into another agent environment
43
+
44
+ `buildTraceAnalysisToolDescriptors()` is the canonical definition of the analyst's seven bounded read operations and does not expose Ax types.
45
+ Each descriptor carries the stable `traces` namespace, function name, description, JSON input schema in `parameters`, and a handler already bound to the supplied `TraceAnalysisStore`.
46
+ `buildTraceAnalystTools()` adapts those descriptors into Ax functions; it does not define a second tool surface.
47
+ The bound handlers wrap custom stores with `createBoundedTraceAnalysisStore()`, so page limits, byte ceilings, not-found errors, and cancellation do not depend on the transport or adapter.
26
48
 
27
49
  ```ts
28
50
  import {
29
- OtlpFileTraceStore,
30
- analyzeTraces,
31
- } from '@tangle-network/agent-eval'
51
+ buildTraceAnalysisToolDescriptors,
52
+ type TraceAnalysisStore,
53
+ } from '@tangle-network/agent-eval/traces'
32
54
 
33
- const abortController = new AbortController()
34
- const result = await analyzeTraces({
35
- question: 'Why did app-runtime holdout runs fail this week?',
36
- }, {
37
- source: new OtlpFileTraceStore({ path: 'traces/otlp.jsonl' }),
38
- ai,
39
- model: 'gpt-4o-2024-11-20',
40
- maxSubqueries: 4,
41
- maxParallelSubqueries: 2,
42
- signal: abortController.signal,
43
- })
55
+ declare const store: TraceAnalysisStore
56
+ declare function qualifyToolName(namespace: string, name: string): string
44
57
 
45
- console.log(result.findings)
58
+ const tools = buildTraceAnalysisToolDescriptors({ store }).map(
59
+ ({ namespace, name, description, parameters, handler }) => ({
60
+ name: qualifyToolName(namespace, name),
61
+ description,
62
+ inputSchema: parameters,
63
+ handler,
64
+ }),
65
+ )
46
66
  ```
47
67
 
48
- Products can pass any `TraceAnalysisStore`; they do not need to use the file store in production.
49
- The analyst runs one Ax executor loop and accepts only an explicit structured `final(task, { report, findings })` result; max-turn fallback text fails loud.
68
+ Map these fields into the host's existing tool transport.
69
+ The host owns namespace encoding; use its existing convention instead of inventing one here.
70
+ Do not copy the schemas or reimplement the handlers in an MCP, Runtime, or provider adapter.
71
+
72
+ `queryTraces.limit`, `viewSpans.span_ids`, and search `max_matches` caps are present in the JSON Schemas and enforced before store calls.
73
+ Invalid arguments throw `TraceAnalysisValidationError` with code `validation`; responses that cannot fit their byte ceiling throw `TraceAnalysisLimitError` with code `limit_exceeded`.
74
+ Search patterns use RE2 syntax, which rejects backreferences and lookaround instead of allowing exponential-time expressions.
75
+ Search results return `hits` and an exact `has_more` flag; they do not invent a total after a capped scan.
76
+ `viewSpans` partitions every requested id across `spans`, `missing_span_ids`, and `omitted_span_ids`; `has_more` is true when omitted ids must be retried.
77
+ Attribute and match text shortening includes a deterministic marker.
78
+ Trace pages set `has_more`, and the overview returns every error cluster or fails explicitly when the configured response limit is too small.
50
79
 
51
80
  ### Analyze captured tool spans in memory
52
81
 
@@ -88,35 +117,442 @@ for (const c of overview.error_clusters) {
88
117
  }
89
118
  ```
90
119
 
91
- See `failureClusters` in [insight-report.md](./insight-report.md) and the
92
- `ErrorCluster` type doc-comments for the field-level contract.
120
+ ## Recursive control integrity (no LLM)
121
+
122
+ `CONTROL_INTEGRITY_ANALYST` checks the existing `SupervisorRunSources` or `SupervisorRunTree` directly.
123
+ It does not define another run format.
124
+ Register it as a custom-input analyst and pass the existing value under its stable id:
125
+
126
+ ```ts
127
+ import {
128
+ AnalystRegistry,
129
+ CONTROL_INTEGRITY_ANALYST,
130
+ } from '@tangle-network/agent-eval/analyst'
131
+ import {
132
+ readLoopsSupervisorRun,
133
+ } from '@tangle-network/agent-eval/supervisor-run'
134
+
135
+ const sources = await readLoopsSupervisorRun(runDir)
136
+ const registry = new AnalystRegistry()
137
+ registry.register(CONTROL_INTEGRITY_ANALYST)
138
+
139
+ const result = await registry.run('run-123', {
140
+ custom: { 'control-integrity': sources },
141
+ })
142
+ ```
143
+
144
+ Pass `SupervisorRunSources` when it is available.
145
+ A `SupervisorRunTree` does not retain raw journal multiplicity or worker request and acknowledgement rows, so tree input explicitly reports those checks as unavailable.
146
+
147
+ The deterministic pass can prove only facts represented by these two existing surfaces.
148
+
149
+ | Question | Current evidence | What the analyst can say |
150
+ |---|---|---|
151
+ | Is every invocation attached to one unambiguous tree? | `rootId`, `rollout_id`, `parent_rollout_id`, `run_id` | Duplicate ids, missing parents, extra parentless roots, cross-run edges, and ancestry cycles are violations with exact field references. |
152
+ | Did invocation roles survive capture? | Explicit journal and `RolloutLine.role` values | The root must remain `supervisor`; non-root roles are consumed as recorded, and workers may spawn workers. |
153
+ | Is the causal order possible? | `outcome.metrics.spawned_at`, `started_at`, `settled_at`, `completed_at`, `finished_at` when present | A child before its parent, a child after its parent closed, or a close before a start is a violation; absent timestamps produce no timing claim. |
154
+ | Did a queued steer reach the worker? | `SupervisorRunSources.workers[].inbox` and `.events` | Requests and acknowledgements are joined by request id, not compared as totals. Missing, malformed, duplicate, or uncorrelated rows make the affected count unavailable. |
155
+ | Can behavior be attributed to an exact profile? | `policy.agent_profile_cell_id` | An absent id is reported as unavailable. |
156
+ | Can action authorship or reasoning be inspected? | `messages[]` | Empty gap rows are reported as unavailable. |
157
+
158
+ An empty finding list means only that no implemented rule fired on the captured fields.
159
+ It does not certify that an agent chose the action, that the action was authorized, that a budget or depth limit was enforced, or that a finding caused a later decision.
160
+ Those claims require upstream action-decision events carrying `action_id`, `actor_rollout_id`, `target_rollout_id`, `action_kind`, `authority_snapshot_id`, requested and granted resource/depth values, the authorization result, and any `finding_id` or evidence references that caused the action.
161
+ Resume integrity additionally requires an explicit prior-session id and resumed-session id rather than a prose summary.
162
+
163
+ Malformed source rows are excluded from structural claims.
164
+ Their count is retained in `SupervisorRunTree.gaps`, so analyzing a projected tree later cannot turn an unreadable parent row into a missing-parent violation.
165
+
166
+ ## Add A Custom Analyst
167
+
168
+ `defineTraceAnalyst()` fills the fixed registry fields.
169
+ The custom function receives the same bounded `TraceAnalysisStore` used by the built-ins.
170
+
171
+ ```ts
172
+ import {
173
+ AnalystRegistry,
174
+ defineTraceAnalyst,
175
+ makeFinding,
176
+ } from '@tangle-network/agent-eval/analyst'
177
+
178
+ const analysts = new AnalystRegistry()
179
+
180
+ analysts.register(defineTraceAnalyst({
181
+ id: 'repeated-tool-errors',
182
+ description: 'Reports the largest repeated tool error cluster.',
183
+ cost: { kind: 'deterministic' },
184
+ async analyze(store) {
185
+ const overview = await store.getOverview({ has_errors: true })
186
+ const cluster = overview.error_clusters[0]
187
+ if (!cluster) return []
188
+
189
+ return [makeFinding({
190
+ analyst_id: 'repeated-tool-errors',
191
+ area: 'tool-use',
192
+ subject: cluster.signature,
193
+ claim: `${cluster.span_count} failed spans share one error`,
194
+ severity: 'high',
195
+ confidence: 1,
196
+ evidence_refs: [{
197
+ kind: 'span',
198
+ uri: `trace://${encodeURIComponent(cluster.exemplar_trace_ids[0])}/span/${encodeURIComponent(cluster.exemplar_span_ids[0])}`,
199
+ excerpt: cluster.status_message_sample,
200
+ }],
201
+ recommended_action: 'Fix the highest-frequency tool error before changing prompts.',
202
+ validation_plan: 'Run fresh cases and require this error signature to disappear.',
203
+ })]
204
+ },
205
+ }))
206
+ ```
207
+
208
+ Use code for exact facts such as exit codes, missing fields, and repeated calls.
209
+ Use model-assisted analysts for semantic questions such as whether a response ignored user intent.
210
+
211
+ ## Measure An Analyst
212
+
213
+ Do not judge an analyst by persuasive prose.
214
+ Label the issue identity and exact evidence locations, then run the same cases through every implementation.
215
+
216
+ ```ts
217
+ import {
218
+ compareAnalystRunners,
219
+ registryBenchmarkRunner,
220
+ renderAnalystBenchmarkMarkdown,
221
+ runAnalystBenchmark,
222
+ traceStoreEvidenceResolver,
223
+ } from '@tangle-network/agent-eval/analyst'
224
+
225
+ const benchmark = await runAnalystBenchmark({
226
+ cases: [{
227
+ id: 'failed-command',
228
+ clusterId: 'incident-42',
229
+ labelState: 'positive',
230
+ input: { traceStore },
231
+ expectedIssues: [{
232
+ id: 'repeated-command',
233
+ subjects: ['failure-mode:repeated-command'],
234
+ evidence: [{ kind: 'span', uri: 'trace://run-1/span/tool-3' }],
235
+ criticalEvidence: [{ kind: 'span', uri: 'trace://run-1/span/tool-1' }],
236
+ }],
237
+ labeledEvidence: [
238
+ { kind: 'span', uri: 'trace://run-1/span/tool-1' },
239
+ { kind: 'span', uri: 'trace://run-1/span/tool-3' },
240
+ ],
241
+ }],
242
+ runners: [registryBenchmarkRunner({ id: 'built-in', registry: analysts })],
243
+ repetitions: 3,
244
+ resolveEvidence: traceStoreEvidenceResolver((input) => input.traceStore),
245
+ benchmark: {
246
+ id: 'failure-localization',
247
+ dataset: {
248
+ id: 'my-team/trace-failures',
249
+ revision: 'git-sha-or-content-digest',
250
+ split: 'test',
251
+ },
252
+ },
253
+ })
254
+
255
+ console.log(renderAnalystBenchmarkMarkdown(benchmark))
256
+ ```
257
+
258
+ The result reports:
259
+
260
+ - issue recall and finding precision,
261
+ - first bad step accuracy,
262
+ - citation coverage, exact source-quote coverage, agreement with labeled locations, and actual location resolution,
263
+ - false positives and failures on trusted-negative cases,
264
+ - predictions and failures on unlabeled cases,
265
+ - matched-label agreement and full-prediction agreement,
266
+ - failed runs,
267
+ - latency, calls, every reported token counter, and known or missing cost,
268
+ - dataset revision, case tags, case metadata, and runner metadata.
269
+
270
+ Use `compareAnalystRunners()` for paired differences between two implementations.
271
+ Repetitions are averaged within each case before comparison.
272
+ Treat its interval as inferential only with at least 20 independent cases.
273
+
274
+ ## Load Public Trace Labels
275
+
276
+ Use the published label adapters with `@tangle-network/traces` or your own trajectory loader.
277
+ Agent Eval does not download datasets or own trace capture.
278
+ Load public data at an immutable commit and record that commit in `benchmark.dataset.revision`.
279
+
280
+ ```ts
281
+ import {
282
+ agentRxBenchmarkCase,
283
+ agentRxPredictionsToFindings,
284
+ codeTraceBenchCase,
285
+ codeTracerPredictionsToFindings,
286
+ } from '@tangle-network/agent-eval/analyst'
287
+ import { otlpTextToTraceAnalysisStore } from '@tangle-network/agent-eval/traces'
288
+ import { chatTrajectoryToSpans, serializeSpans } from '@tangle-network/traces'
289
+
290
+ const codeSpans = chatTrajectoryToSpans(codeTraceTrajectory, {
291
+ traceId: codeTraceRow.traj_id,
292
+ })
293
+ const codeCase = codeTraceBenchCase(codeTraceRow, {
294
+ traceStore: otlpTextToTraceAnalysisStore(serializeSpans(codeSpans)),
295
+ })
296
+
297
+ const agentRxSpans = chatTrajectoryToSpans(agentRxMessages, {
298
+ traceId: String(agentRxRow.trajectory_id),
299
+ stepMode: 'message',
300
+ })
301
+ const rootCauseCase = agentRxBenchmarkCase(agentRxRow, {
302
+ traceStore: otlpTextToTraceAnalysisStore(serializeSpans(agentRxSpans)),
303
+ }, {
304
+ stepCount: agentRxMessages.length,
305
+ })
306
+
307
+ const codeTracerFindings = codeTracerPredictionsToFindings(
308
+ codeTraceRow.traj_id,
309
+ codetracerLabels,
310
+ { stepCount: codeTraceRow.step_count },
311
+ )
312
+ const agentRxFindings = agentRxPredictionsToFindings(
313
+ agentRxRow.trajectory_id,
314
+ agentRxJudgeOutput,
315
+ { stepCount: agentRxMessages.length },
316
+ )
317
+ ```
318
+
319
+ `codeTraceBenchCase()` accepts the public [CodeTraceBench](https://huggingface.co/datasets/NJU-LINK/CodeTraceBench) JSONL format.
320
+ It scores the published incorrect-step task by default, including clean trajectories.
321
+ Pass `labelSet: 'incorrect-and-unuseful'` to both the case and prediction adapters only for an explicitly combined experiment.
322
+ Every cited step is checked against `step_count`.
323
+
324
+ `agentRxBenchmarkCase()` accepts the public [AgentRx](https://huggingface.co/datasets/microsoft/AgentRx) label format.
325
+ AgentRx category quality and root-step accuracy are scored independently.
326
+ `traceAnalystQualityJudge` averages them when a root-step label exists.
327
+ Pass `target: 'all-failures'` only when the analyst is designed to identify every annotated failure.
328
+
329
+ Both adapters emit `trace://<id>/span/step-<n>` evidence by default.
330
+ `@tangle-network/traces` uses the same IDs when converting chat trajectories.
331
+ Pass `stepUri` when your trace store uses another URI scheme.
332
+ `codeTracerPredictionsToFindings()` and `agentRxPredictionsToFindings()` translate the maintained upstream engines' native outputs into the same evidence and category shape.
333
+ The CodeTracer adapter accepts the published `stage_id` format and the flat or grouped step-label formats emitted by CodeTracer 0.2.
334
+ AgentRx `Report.to_dict()` judge votes reduce to the upstream majority failure type and Python-rounded mean step, and direct `failures` arrays use the same reduction.
335
+ `failure_case: 0` produces no finding, which scores as a missed root cause on AgentRx's failed trajectories.
336
+ External runners can return `observedLatencyMs`, `usage`, `metadata`, and `error` together.
337
+ This records an upstream failure without discarding work already performed.
338
+ Set `observedLatencyMs: null` when an imported run did not record duration.
339
+ The report keeps it unknown instead of timing the import code.
340
+
341
+ ## Run A Real-Model Public Benchmark
342
+
343
+ `agent-eval analyst-benchmark` runs the existing label adapters and `runAnalystBenchmark()` with two runners: an empty-finding baseline and a benchmark-specific model analyst.
344
+ The CodeTraceBench runner emits one prediction per incorrect step, including wrong actions that the agent later recovers from.
345
+ Final task success is evidence about the final state, not proof that every earlier action was correct.
346
+ Unuseful but correct exploration is a separate CodeTraceBench label and is not scored by the default run.
347
+ The AgentRx runner emits one taxonomy label and one root-cause step.
348
+ The generic `failure-mode` analyst is not used because its `failure-mode` area does not match either public task.
349
+
350
+ Convert CodeTraceBench trajectories with the maintained importer.
351
+ It writes one OTLP JSONL file per trajectory, preserves assistant step ids, and produces a receipt with source and output hashes.
352
+ Each label row's `traj_id` or `trajectory_id` must exactly equal the OTLP `trace_id`.
353
+ Use a domain-qualified ID when an upstream dataset reuses local trajectory numbers.
354
+ Keep the extracted CodeTraceBench artifact tree intact because each row's `source_relpath` locates its final test output.
355
+
356
+ ```bash
357
+ traces import-codetracebench \
358
+ .artifacts/bench_manifest.verified.jsonl \
359
+ --trajectory-dir .artifacts/codetrace-normalized \
360
+ --out .artifacts/codetrace-otlp \
361
+ --revision aa213b84ffb6690fc37ca15766d6ca174ec36d4d \
362
+ --concurrency 8
363
+ ```
364
+
365
+ Run a bounded comparison through any OpenAI-compatible endpoint.
366
+ The key is read from the named environment variable and is not written to the result.
367
+
368
+ ```bash
369
+ export CLI_BRIDGE_BEARER="<read from the running bridge environment>"
370
+
371
+ agent-eval analyst-benchmark \
372
+ --dataset codetracebench \
373
+ --labels .artifacts/bench_manifest.verified.jsonl \
374
+ --trace-dir .artifacts/codetrace-otlp \
375
+ --artifact-dir .artifacts/codetrace-extracted \
376
+ --out .artifacts/codetrace-glm52 \
377
+ --revision aa213b84ffb6690fc37ca15766d6ca174ec36d4d \
378
+ --split verified \
379
+ --base-url http://127.0.0.1:3355/v1 \
380
+ --api-key-env CLI_BRIDGE_BEARER \
381
+ --model opencode/zai-coding-plan/glm-5.2 \
382
+ --limit 20 \
383
+ --seed 7 \
384
+ --concurrency 4 \
385
+ --max-cost-usd 5
386
+ ```
387
+
388
+ The trace directory may use any filenames, but every JSONL file must contain exactly one trace.
389
+ For CodeTraceBench, the artifact loader reads available final test output and structured result files.
390
+ It adds raw final-test text and one parsed pass, fail, or unavailable outcome as searchable `EVALUATOR` spans on the same case.
391
+ Raw result JSON is hashed and parsed but is not sent to the model.
392
+ `--artifact-dir` may be a shared extraction root with one directory per `traj_id`, or the extraction root for one archive.
393
+ It prefers `panes/post-test.txt` over the duplicate `sessions/tests.log`.
394
+ It also recognizes `test_output.txt`, `results.json`, `result.json`, `report.json`, `*_result.json`, and `*_metrics.json`.
395
+ Each discovered file records its role, path, byte count, and SHA-256.
396
+ Files exposed as spans also record their span ids.
397
+ Missing evidence roles are explicit.
398
+ Known Terminal-Bench, SWE-Bench, and SWE-Multi result formats become one explicit passed or failed outcome span.
399
+ A missing or unparseable result becomes `unavailable`; it is never inferred from the trajectory or raw test text.
400
+ Raw test output is optional because some public cases retain only structured results.
401
+
402
+ Before the first model call, the command also checks that every selected label has a matching `step-<n>` span.
403
+ It refuses a missing dataset revision, implicit all-case run, duplicate trajectory id, multi-trace file, missing step, oversized evidence, or existing `result.json`.
404
+ The model selects positive integer assistant step ids.
405
+ The runner constructs each canonical trace URI and exact action excerpt from the selected span.
406
+ A missing, non-assistant, or empty step fails that model run while preserving its raw output and usage.
407
+ The command consumes already-downloaded labels, normalized traces, and extracted artifacts.
408
+ Dataset download, archive extraction, and trajectory conversion remain separate import steps.
409
+
410
+ The output directory contains:
411
+
412
+ - `manifest.json` with the immutable dataset, model, case, and input identity.
413
+ - `initialization-complete.json` written only after every initial file is durable.
414
+ - `observations.jsonl` with one fsynced, hash-chained row per completed case and runner.
415
+ - `cost-ledger.jsonl` with durable run-wide model reservations and receipts.
416
+ - `model-responses/` with one strict, content-hashed response and receipt per paid call for crash recovery.
417
+ - `result.json` with every observation, summary metric, comparison, error, latency, token counter, measured or estimated cost, explicit unknown cost, selected case id, source digest, dependency-lock digest, artifact digest, analyst protocol digest, implementation digest, and case distribution.
418
+ - `report.md` with the same run and selection distribution rendered for review.
419
+ - `run.local.json` with machine-local paths, endpoint, and command, kept out of the shareable result.
420
+
421
+ When active calls temporarily hold the remaining money limit, later calls wait for their final receipts.
422
+ The command rejects new work only when committed spend plus the next enforced maximum cannot fit.
93
423
 
94
- ## Required Trace Shape
424
+ Resume only the exact same run after an interruption:
95
425
 
96
- Every serious product run should include:
426
+ ```bash
427
+ agent-eval analyst-benchmark <the same flags> --resume
428
+ ```
429
+
430
+ Resume rejects changed labels, traces, artifacts, model settings, case selection, local paths, or endpoint.
431
+ It runs only missing case and runner pairs.
432
+ If the provider response was saved before the process stopped, resume settles that exact response without another provider call.
433
+ If the call stopped before a response was saved, resume reuses the same provider request id.
434
+ An already complete run is read and checked without another model call.
435
+
436
+ The command exits `2` when any model analyst fails.
437
+ Its failed row still records latency, calls, available token counters, known spend, and the analyst error.
438
+ See the [32-case GLM-5.2 reference run](https://github.com/tangle-network/agent-eval/tree/main/benchmarks/trace-analysis/codetracebench-glm52-20260730) for a complete measured result and its stated limits.
439
+
440
+ `--limit` uses deterministic hash selection, not stratified sampling.
441
+ Limited runs are marked `representativeOfInput: false`.
442
+ The result compares source and selected distributions for label class, agent, model, difficulty, and solved state.
443
+ CodeTraceBench label classes distinguish `positive`, `trusted-negative`, `unlabeled-failure`, and `unlabeled-unknown`.
444
+ Micro precision, recall, and F1 pool all labeled steps and predictions.
445
+ Macro precision, recall, and F1 average per-case scores over issue-bearing cases, matching step-localization papers that report per-trajectory means.
446
+ The all-row result remains intact for comparison with published work.
447
+ The additional calibrated view measures labeled positives against solved, label-empty negatives and reports failed, label-empty rows separately instead of calling them clean.
448
+ The report separately counts final-result files and passed, failed, or unavailable outcomes.
449
+ Only a full census of the supplied input is marked representative.
450
+
451
+ AgentRx uses the same command with `--dataset agentrx` after obtaining its contact-gated label and trajectory files.
452
+
453
+ ## Use Upstream Scorers
454
+
455
+ Agent Eval adapts upstream evaluators instead of copying them.
456
+
457
+ ```ts
458
+ import { createEvaluator } from '@arizeai/phoenix-evals'
459
+ import { ExactMatch } from 'autoevals'
460
+ import {
461
+ autoevalsScorerJudge,
462
+ phoenixEvaluatorJudge,
463
+ } from '@tangle-network/agent-eval/campaign'
464
+
465
+ const phoenix = createEvaluator(
466
+ ({ output, expected }) => output === expected ? 1 : 0,
467
+ { name: 'exact-match', kind: 'CODE', telemetry: { isEnabled: false } },
468
+ )
469
+
470
+ const phoenixJudge = phoenixEvaluatorJudge(phoenix, {
471
+ mapInput: ({ artifact, scenario }) => ({ output: artifact, expected: scenario.expected }),
472
+ })
473
+
474
+ const autoevalsJudge = autoevalsScorerJudge(ExactMatch, {
475
+ name: 'exact-match',
476
+ kind: 'CODE',
477
+ mapInput: ({ artifact, scenario }) => ({ output: artifact, expected: scenario.expected }),
478
+ })
479
+ ```
480
+
481
+ These adapters do not install either upstream package for consumers.
482
+ Install only the scorer package you use.
483
+ Missing or non-finite scores throw instead of becoming passes.
484
+ Phoenix evaluators marked `MINIMIZE` or `NEUTRAL` require `toComposite` so candidate selection never assumes the wrong direction.
485
+ Mark model-backed evaluators as `kind: 'LLM'` and provide `paidCall` with the model and a receipt mapper.
486
+ The campaign then passes its cancellation signal and cost ledger through the adapter.
487
+ An LLM evaluator is rejected before execution when either cost capture or the campaign ledger is missing.
488
+
489
+ ## Turn Reviewed Findings Into Eval Data
490
+
491
+ Generated findings can populate a review queue.
492
+ They cannot promote themselves into learning data.
493
+
494
+ ```ts
495
+ import {
496
+ analystFindingDigest,
497
+ analystRunDigest,
498
+ analystRunToFeedbackTrajectory,
499
+ analystRunToReviewRequests,
500
+ } from '@tangle-network/agent-eval'
501
+
502
+ const runDigest = analystRunDigest(result)
503
+ const requests = analystRunToReviewRequests(result)
504
+ await reviewQueue.add(requests)
505
+
506
+ declare const acceptedFindingIds: ReadonlySet<string>
507
+
508
+ const trajectory = analystRunToFeedbackTrajectory(result, {
509
+ task: { intent: 'Find why the command failed.' },
510
+ reviewRequests: requests,
511
+ reviewDecisions: [
512
+ ...result.findings.map((finding) => ({
513
+ runDigest,
514
+ findingId: finding.finding_id,
515
+ findingDigest: analystFindingDigest(finding),
516
+ verdict: acceptedFindingIds.has(finding.finding_id) ? 'confirmed' as const : 'rejected' as const,
517
+ source: 'user' as const,
518
+ reviewerId: 'reviewer-42',
519
+ reviewId: 'trace-review-918',
520
+ reason: 'Reviewed against the cited span.',
521
+ decidedAt: new Date().toISOString(),
522
+ })),
523
+ {
524
+ runDigest,
525
+ verdict: 'completeness_assessed',
526
+ missedIssues: [],
527
+ source: 'user',
528
+ reviewerId: 'reviewer-42',
529
+ reviewId: 'trace-review-918',
530
+ reason: 'Reviewed the full run for omitted findings.',
531
+ decidedAt: new Date().toISOString(),
532
+ },
533
+ ],
534
+ trace: { artifactUri: 'traces.otlp.jsonl', traceIds: ['run-1'] },
535
+ })
536
+ ```
97
537
 
98
- - `runId`, `projectId`, `scenarioId`, `variantId`, and `layer`
99
- - commit, prompt hash, config hash, model fingerprint, and dataset version
100
- - LLM spans with model, inputs, outputs, token counts, and cost
101
- - tool/integration spans with arguments, result summaries, and error codes
102
- - retrieval spans with query, source ids, hit scores, and freshness metadata
103
- - sandbox/build/test/deploy spans with exit codes and log artifacts
104
- - custom events for knowledge readiness and integration gates
105
- - final run outcome with pass/score/failure class
538
+ `analystRunToFeedbackTrajectory()` stores review requests separately from labels.
539
+ It can archive an unreviewed run.
540
+ `feedbackTrajectoryToOptimizerRow()` requires every decision to match the complete run digest, every finding decision to match the finding digest, and one independent completeness assessment.
541
+ Its score is F1 over confirmed findings and independently identified misses.
542
+ Generic labels and run-level outcomes do not satisfy these requirements.
106
543
 
107
- Do not put secrets, raw OAuth tokens, or unredacted PII in traces.
544
+ ## Required Trace Data
108
545
 
109
- ## Product Loop
546
+ Useful analysis needs:
110
547
 
111
- The product loop should not treat traces as a separate debug dump. The intended
112
- path is:
548
+ - stable run, trace, and span IDs,
549
+ - parent-child links and ordered timestamps,
550
+ - model, prompt, and configuration identity,
551
+ - complete tool names, arguments, results, and error codes,
552
+ - token, cost, and latency data when available,
553
+ - retrieval source IDs and scores when retrieval is involved,
554
+ - final environment outcomes such as tests, task completion, or policy blocks.
113
555
 
114
- 1. Wrap the real workflow in `runAgentControlLoop` or the product runtime.
115
- 2. Emit canonical spans/events while the user task runs.
116
- 3. Convert the completed run to `FeedbackTrajectory` for replay.
117
- 4. Convert promotion-grade runs to `RunRecord` with `controlRunToRunRecord`.
118
- 5. Run TraceAnalyst over failure-heavy trace sets.
119
- 6. Feed findings into `ActionableSideInfo`, failure clusters, and release
120
- reports.
556
+ Do not include secrets, raw OAuth tokens, or unredacted personal data.
121
557
 
122
- That makes normal product usage become eval data instead of isolated logs.
558
+ Use [`@tangle-network/traces`](https://github.com/tangle-network/traces) to normalize coding-agent sessions, run HALO as an external report engine, or run Hodoscope as a behavior-discovery engine.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.136.0",
3
+ "version": "0.138.0",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {
@@ -150,7 +150,7 @@
150
150
  "access": "public"
151
151
  },
152
152
  "scripts": {
153
- "build": "tsdown && pnpm openapi",
153
+ "build": "pnpm check:analyst-benchmark && tsdown && pnpm openapi",
154
154
  "dev": "tsdown --watch",
155
155
  "prepare": "husky",
156
156
  "prepublishOnly": "pnpm build",
@@ -161,8 +161,9 @@
161
161
  "lint": "biome check src",
162
162
  "format": "biome format --write src",
163
163
  "check:skill": "node scripts/check-skill.mjs",
164
+ "check:analyst-benchmark": "node scripts/check-analyst-benchmark-implementation.mjs",
164
165
  "openapi": "node dist/cli.js openapi --out dist/openapi.json",
165
- "verify:package": "pnpm run check:skill && publint && attw --pack --profile esm-only . && node scripts/verify-package-exports.mjs"
166
+ "verify:package": "pnpm check:analyst-benchmark && pnpm run check:skill && publint && attw --pack --profile esm-only . && node scripts/verify-package-exports.mjs"
166
167
  },
167
168
  "dependencies": {
168
169
  "@asteasolutions/zod-to-openapi": "^9.1.0",
@@ -171,12 +172,16 @@
171
172
  "@tangle-network/agent-core": "0.4.28",
172
173
  "@tangle-network/agent-interface": "0.39.0",
173
174
  "hono": "^4.12.32",
175
+ "linear-sum-assignment": "1.0.9",
176
+ "re2js": "2.8.6",
174
177
  "zod": "^4.4.3"
175
178
  },
176
179
  "devDependencies": {
177
180
  "@arethetypeswrong/cli": "^0.18.5",
181
+ "@arizeai/phoenix-evals": "^2.1.0",
178
182
  "@biomejs/biome": "^2.5.5",
179
183
  "@types/node": "^26.1.1",
184
+ "autoevals": "^0.3.0",
180
185
  "esbuild": "^0.28.1",
181
186
  "fast-check": "^4.9.0",
182
187
  "husky": "^9.1.7",
@@ -197,6 +202,7 @@
197
202
  "vite"
198
203
  ],
199
204
  "overrides": {
205
+ "@arizeai/openinference-core>@opentelemetry/core": "^2.10.0",
200
206
  "esbuild@>=0.27.3 <0.28.1": "^0.28.1",
201
207
  "postcss@<8.5.18": "^8.5.18",
202
208
  "vite@>=7.0.0 <=7.3.4": "^7.3.5",