@tangle-network/agent-eval 0.136.0 → 0.138.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (231) hide show
  1. package/CHANGELOG.md +86 -1
  2. package/README.md +37 -2
  3. package/dist/{agent-profile-cell-OhuTee9n.js → agent-profile-cell-CbfBm2g6.js} +2 -2
  4. package/dist/{agent-profile-cell-OhuTee9n.js.map → agent-profile-cell-CbfBm2g6.js.map} +1 -1
  5. package/dist/{agent-profile-cell-CCm3l2v2.d.ts → agent-profile-cell-Cw0PVwDr.d.ts} +2 -2
  6. package/dist/{agent-profile-cell-CCm3l2v2.d.ts.map → agent-profile-cell-Cw0PVwDr.d.ts.map} +1 -1
  7. package/dist/analyst/index.d.ts +574 -18
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +25 -5
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/{analyze-runs-jjCmF8pU.js → analyze-runs-BScZvqMV.js} +6 -6
  12. package/dist/{analyze-runs-jjCmF8pU.js.map → analyze-runs-BScZvqMV.js.map} +1 -1
  13. package/dist/{analyze-runs-Cda5Xkj1.d.ts → analyze-runs-CPYxfPWT.d.ts} +6 -6
  14. package/dist/{analyze-runs-Cda5Xkj1.d.ts.map → analyze-runs-CPYxfPWT.d.ts.map} +1 -1
  15. package/dist/{baseline-BUeFcgrn.js → baseline-C-GocmIW.js} +2 -2
  16. package/dist/{baseline-BUeFcgrn.js.map → baseline-C-GocmIW.js.map} +1 -1
  17. package/dist/benchmark-D8dkki-J.js +554 -0
  18. package/dist/benchmark-D8dkki-J.js.map +1 -0
  19. package/dist/benchmark-DlQgU_XI.d.ts +236 -0
  20. package/dist/benchmark-DlQgU_XI.d.ts.map +1 -0
  21. package/dist/benchmark-command-CMqVqReF.js +4332 -0
  22. package/dist/benchmark-command-CMqVqReF.js.map +1 -0
  23. package/dist/benchmarks/index.d.ts +1 -1
  24. package/dist/benchmarks/index.js +1 -1
  25. package/dist/{benchmarks-Dfgm9ts5.js → benchmarks-BJ_xK5rQ.js} +4 -3
  26. package/dist/{benchmarks-Dfgm9ts5.js.map → benchmarks-BJ_xK5rQ.js.map} +1 -1
  27. package/dist/builder-eval/index.js +2 -2
  28. package/dist/campaign/index.d.ts +6 -6
  29. package/dist/campaign/index.js +4 -4
  30. package/dist/{campaign-Dz8uQnhC.js → campaign-BIBS-NHV.js} +219 -78
  31. package/dist/campaign-BIBS-NHV.js.map +1 -0
  32. package/dist/cli.js +9 -2
  33. package/dist/cli.js.map +1 -1
  34. package/dist/client-BwPKohkJ.d.ts +202 -0
  35. package/dist/client-BwPKohkJ.d.ts.map +1 -0
  36. package/dist/completion-verifier-B4-IMYcS.d.ts +240 -0
  37. package/dist/completion-verifier-B4-IMYcS.d.ts.map +1 -0
  38. package/dist/contract/index.d.ts +10 -9
  39. package/dist/contract/index.d.ts.map +1 -1
  40. package/dist/contract/index.js +13 -13
  41. package/dist/control.d.ts +2 -2
  42. package/dist/control.js +1 -1
  43. package/dist/{cost-ledger-fGS_u_O1.d.ts → cost-ledger-B1D3COAc.d.ts} +6 -5
  44. package/dist/{cost-ledger-fGS_u_O1.d.ts.map → cost-ledger-B1D3COAc.d.ts.map} +1 -1
  45. package/dist/{cost-ledger-DHAjwNj7.js → cost-ledger-CHDLA0Ss.js} +91 -46
  46. package/dist/cost-ledger-CHDLA0Ss.js.map +1 -0
  47. package/dist/{dataset-BvtnC8Dc.d.ts → dataset-v_Y5902-.d.ts} +2 -2
  48. package/dist/{dataset-BvtnC8Dc.d.ts.map → dataset-v_Y5902-.d.ts.map} +1 -1
  49. package/dist/{default-registry-Brxr728w.d.ts → default-registry-PUhIVRWz.d.ts} +77 -138
  50. package/dist/default-registry-PUhIVRWz.d.ts.map +1 -0
  51. package/dist/{default-registry-CHmdy2An.js → default-registry-lp5R0lve.js} +1617 -291
  52. package/dist/default-registry-lp5R0lve.js.map +1 -0
  53. package/dist/{errors-8YnH8WlF.js → errors-D-LKuDhb.js} +8 -2
  54. package/dist/errors-D-LKuDhb.js.map +1 -0
  55. package/dist/{errors-CEk209JS.d.ts → errors-DkfjIDvD.d.ts} +9 -3
  56. package/dist/errors-DkfjIDvD.d.ts.map +1 -0
  57. package/dist/{eval-campaign-Cc8WZJ6b.js → eval-campaign-9MozgKL7.js} +6 -6
  58. package/dist/{eval-campaign-Cc8WZJ6b.js.map → eval-campaign-9MozgKL7.js.map} +1 -1
  59. package/dist/exact-types-Dpw2LeHA.d.ts +234 -0
  60. package/dist/exact-types-Dpw2LeHA.d.ts.map +1 -0
  61. package/dist/{extract-usage-DIQpN-ww.js → extract-usage-CS391dOE.js} +3 -3
  62. package/dist/{extract-usage-DIQpN-ww.js.map → extract-usage-CS391dOE.js.map} +1 -1
  63. package/dist/{feedback-trajectory-CVaeREXV.d.ts → feedback-trajectory-CoNep7rl.d.ts} +91 -3
  64. package/dist/feedback-trajectory-CoNep7rl.d.ts.map +1 -0
  65. package/dist/fuzz.d.ts +1 -1
  66. package/dist/fuzz.js +2 -2
  67. package/dist/hosted/index.d.ts +3 -2
  68. package/dist/hosted/index.d.ts.map +1 -1
  69. package/dist/{index-AbhwHp0V.d.ts → index-B2-IxCMB.d.ts} +2 -2
  70. package/dist/{index-AbhwHp0V.d.ts.map → index-B2-IxCMB.d.ts.map} +1 -1
  71. package/dist/index-BipJlj-C.d.ts +316 -0
  72. package/dist/index-BipJlj-C.d.ts.map +1 -0
  73. package/dist/{index-CQsJcqch.d.ts → index-CjVYlVBK.d.ts} +5 -5
  74. package/dist/{index-CQsJcqch.d.ts.map → index-CjVYlVBK.d.ts.map} +1 -1
  75. package/dist/{index-B4Fjfo5U.d.ts → index-D0cxAdaV.d.ts} +89 -317
  76. package/dist/index-D0cxAdaV.d.ts.map +1 -0
  77. package/dist/{index-DuhJaaiH.d.ts → index-DEb46kc6.d.ts} +2 -2
  78. package/dist/index-DEb46kc6.d.ts.map +1 -0
  79. package/dist/{index-C2fkZhv_.d.ts → index-sMN_hI4E.d.ts} +3 -3
  80. package/dist/{index-C2fkZhv_.d.ts.map → index-sMN_hI4E.d.ts.map} +1 -1
  81. package/dist/index.d.ts +30 -70
  82. package/dist/index.d.ts.map +1 -1
  83. package/dist/index.js +175 -35
  84. package/dist/index.js.map +1 -1
  85. package/dist/{client-DcvgkaZi.d.ts → insight-report-CXd8VBDR.d.ts} +5 -203
  86. package/dist/insight-report-CXd8VBDR.d.ts.map +1 -0
  87. package/dist/{integrity-rmVhXWA7.d.ts → integrity-B-MLFz0I.d.ts} +3 -3
  88. package/dist/{integrity-rmVhXWA7.d.ts.map → integrity-B-MLFz0I.d.ts.map} +1 -1
  89. package/dist/integrity-CCXTftiL.js +1360 -0
  90. package/dist/integrity-CCXTftiL.js.map +1 -0
  91. package/dist/{integrity-BzRbCHzi.js → integrity-fdt8XPAv.js} +2 -2
  92. package/dist/{integrity-BzRbCHzi.js.map → integrity-fdt8XPAv.js.map} +1 -1
  93. package/dist/ledger-core/index.d.ts +1 -1
  94. package/dist/ledger-core/index.js +1 -1
  95. package/dist/{ledger-core-DAKFKRzi.js → ledger-core-C0Yx1I14.js} +303 -110
  96. package/dist/ledger-core-C0Yx1I14.js.map +1 -0
  97. package/dist/{llm-client-DHx8pzyJ.js → llm-client-Cj3c7PEm.js} +6 -6
  98. package/dist/llm-client-Cj3c7PEm.js.map +1 -0
  99. package/dist/meta-eval/index.d.ts +2 -2
  100. package/dist/meta-eval/index.js +3 -3
  101. package/dist/{mint-DyRUc9k6.js → mint-Ctwk079K.js} +4 -4
  102. package/dist/{mint-DyRUc9k6.js.map → mint-Ctwk079K.js.map} +1 -1
  103. package/dist/multishot/index.d.ts +2 -2
  104. package/dist/openapi.json +1 -1
  105. package/dist/{paired-arms-BbFKrAU-.js → paired-arms-iZ08VFMN.js} +3 -3
  106. package/dist/{paired-arms-BbFKrAU-.js.map → paired-arms-iZ08VFMN.js.map} +1 -1
  107. package/dist/pipelines/index.js +2 -2
  108. package/dist/profile-cell.d.ts +1 -1
  109. package/dist/profile-cell.js +1 -1
  110. package/dist/{proposal-findings-DCawte-y.js → proposal-findings-2GIUo1et.js} +2 -68
  111. package/dist/proposal-findings-2GIUo1et.js.map +1 -0
  112. package/dist/{propose-review-control-SQ-n9-We.js → propose-review-control-DLXz4FCX.js} +2 -2
  113. package/dist/{propose-review-control-SQ-n9-We.js.map → propose-review-control-DLXz4FCX.js.map} +1 -1
  114. package/dist/registry-C4yJTza7.d.ts +178 -0
  115. package/dist/registry-C4yJTza7.d.ts.map +1 -0
  116. package/dist/{release-report-DooPguBc.js → release-report-B5XPBvAU.js} +4 -4
  117. package/dist/{release-report-DooPguBc.js.map → release-report-B5XPBvAU.js.map} +1 -1
  118. package/dist/{release-report-DpBxGGI1.d.ts → release-report-CoyvyLBs.d.ts} +4 -4
  119. package/dist/{release-report-DpBxGGI1.d.ts.map → release-report-CoyvyLBs.d.ts.map} +1 -1
  120. package/dist/{replay-C6wRg47C.js → replay-Cb-4Vf0k.js} +249 -8
  121. package/dist/replay-Cb-4Vf0k.js.map +1 -0
  122. package/dist/{replay-BRfMIs81.d.ts → replay-DbIYwso6.d.ts} +227 -52
  123. package/dist/replay-DbIYwso6.d.ts.map +1 -0
  124. package/dist/reporting.d.ts +4 -4
  125. package/dist/reporting.js +4 -4
  126. package/dist/{researcher-Doo95b50.d.ts → researcher-BCeOEjtR.d.ts} +6 -7
  127. package/dist/researcher-BCeOEjtR.d.ts.map +1 -0
  128. package/dist/{reward-hacking-a-kYs0-i.js → reward-hacking-GyN0kMd8.js} +3 -3
  129. package/dist/{reward-hacking-a-kYs0-i.js.map → reward-hacking-GyN0kMd8.js.map} +1 -1
  130. package/dist/{reward-hacking-D-QqXvg-.d.ts → reward-hacking-sE2l_NV6.d.ts} +2 -2
  131. package/dist/{reward-hacking-D-QqXvg-.d.ts.map → reward-hacking-sE2l_NV6.d.ts.map} +1 -1
  132. package/dist/rl.d.ts +6 -6
  133. package/dist/rl.js +9 -9
  134. package/dist/rollout/index.d.ts +1 -1
  135. package/dist/rollout/index.js +3 -3
  136. package/dist/{rollout-DLSUIWLu.js → rollout-DQFl0UXA.js} +2 -2
  137. package/dist/{rollout-DLSUIWLu.js.map → rollout-DQFl0UXA.js.map} +1 -1
  138. package/dist/{rubric-predictive-validity-BJf-8ejY.js → rubric-predictive-validity-BRR632r1.js} +2 -2
  139. package/dist/{rubric-predictive-validity-BJf-8ejY.js.map → rubric-predictive-validity-BRR632r1.js.map} +1 -1
  140. package/dist/{rubric-predictive-validity-C1dCLcvb.d.ts → rubric-predictive-validity-w2klGv1u.d.ts} +2 -2
  141. package/dist/{rubric-predictive-validity-C1dCLcvb.d.ts.map → rubric-predictive-validity-w2klGv1u.d.ts.map} +1 -1
  142. package/dist/{run-evidence-DokQtX0-.d.ts → run-evidence-CbE0A8Xg.d.ts} +3 -3
  143. package/dist/{run-evidence-DokQtX0-.d.ts.map → run-evidence-CbE0A8Xg.d.ts.map} +1 -1
  144. package/dist/{run-record-DcObtIGh.d.ts → run-record-DwHMk1Ai.d.ts} +4 -4
  145. package/dist/{run-record-DcObtIGh.d.ts.map → run-record-DwHMk1Ai.d.ts.map} +1 -1
  146. package/dist/{run-record-BIwU2wdV.js → run-record-vRgqWmJw.js} +3 -3
  147. package/dist/{run-record-BIwU2wdV.js.map → run-record-vRgqWmJw.js.map} +1 -1
  148. package/dist/{semantic-concept-judge-Btozx3Vc.js → semantic-concept-judge-DYXDPZW0.js} +12 -6
  149. package/dist/semantic-concept-judge-DYXDPZW0.js.map +1 -0
  150. package/dist/{server-Bz3WQJs6.js → server-DLEvyW2z.js} +3 -3
  151. package/dist/{server-Bz3WQJs6.js.map → server-DLEvyW2z.js.map} +1 -1
  152. package/dist/single-run-lock-D_bS5xhj.js +318 -0
  153. package/dist/single-run-lock-D_bS5xhj.js.map +1 -0
  154. package/dist/{skill-usage-BDQVPIG1.d.ts → skill-usage-Bv3G4VkA.d.ts} +36 -48
  155. package/dist/skill-usage-Bv3G4VkA.d.ts.map +1 -0
  156. package/dist/{skillopt-optimization-method-0UmPD6aP.js → skillopt-optimization-method-CjKMZy0d.js} +10 -185
  157. package/dist/skillopt-optimization-method-CjKMZy0d.js.map +1 -0
  158. package/dist/{skillopt-optimization-method-CwSYkv35.d.ts → skillopt-optimization-method-CzfnA8O-.d.ts} +11 -12
  159. package/dist/skillopt-optimization-method-CzfnA8O-.d.ts.map +1 -0
  160. package/dist/{statistics-CnGCLLqc.js → statistics-ByxzSiOM.js} +2 -2
  161. package/dist/{statistics-CnGCLLqc.js.map → statistics-ByxzSiOM.js.map} +1 -1
  162. package/dist/{statistics-CKOqre5S.d.ts → statistics-mf70aXKp.d.ts} +2 -2
  163. package/dist/{statistics-CKOqre5S.d.ts.map → statistics-mf70aXKp.d.ts.map} +1 -1
  164. package/dist/store-otlp-BenKynPE.js +1688 -0
  165. package/dist/store-otlp-BenKynPE.js.map +1 -0
  166. package/dist/{summary-report-BEk8OFLs.js → summary-report-9A5y7EsK.js} +4 -4
  167. package/dist/{summary-report-BEk8OFLs.js.map → summary-report-9A5y7EsK.js.map} +1 -1
  168. package/dist/{summary-report-CPMINBqs.d.ts → summary-report-BKinV4yD.d.ts} +3 -3
  169. package/dist/{summary-report-CPMINBqs.d.ts.map → summary-report-BKinV4yD.d.ts.map} +1 -1
  170. package/dist/supervisor-run/index.d.ts +3 -2
  171. package/dist/supervisor-run/index.js +3 -2
  172. package/dist/{supervisor-run-Dr5HnTup.js → supervisor-run-B2EWUmQY.js} +28 -454
  173. package/dist/supervisor-run-B2EWUmQY.js.map +1 -0
  174. package/dist/{test-graded-scenario-BsqWLmPt.js → test-graded-scenario-JHcKQNpq.js} +2 -2
  175. package/dist/{test-graded-scenario-BsqWLmPt.js.map → test-graded-scenario-JHcKQNpq.js.map} +1 -1
  176. package/dist/tools-DZGdROtG.js +255 -0
  177. package/dist/tools-DZGdROtG.js.map +1 -0
  178. package/dist/traces.d.ts +6 -7
  179. package/dist/traces.js +6 -6
  180. package/dist/types-5q2T25iW.d.ts +804 -0
  181. package/dist/types-5q2T25iW.d.ts.map +1 -0
  182. package/dist/{types-DVjczBM9.d.ts → types-BtJhn8v6.d.ts} +260 -6
  183. package/dist/types-BtJhn8v6.d.ts.map +1 -0
  184. package/dist/{index-CyC1BTmn.d.ts → types-Dea6tiVI.d.ts} +16 -238
  185. package/dist/types-Dea6tiVI.d.ts.map +1 -0
  186. package/dist/{types-DiWLru6Z.d.ts → types-zFYez3PK.d.ts} +5 -5
  187. package/dist/{types-DiWLru6Z.d.ts.map → types-zFYez3PK.d.ts.map} +1 -1
  188. package/dist/wire/index.d.ts +3 -3
  189. package/dist/wire/index.js +1 -1
  190. package/docs/feedback-trajectories.md +100 -1
  191. package/docs/trace-analysis.md +494 -58
  192. package/package.json +9 -3
  193. package/dist/analyst-BkTS3C58.d.ts +0 -89
  194. package/dist/analyst-BkTS3C58.d.ts.map +0 -1
  195. package/dist/analyst-j5je5J7c.js +0 -152
  196. package/dist/analyst-j5je5J7c.js.map +0 -1
  197. package/dist/campaign-Dz8uQnhC.js.map +0 -1
  198. package/dist/client-DcvgkaZi.d.ts.map +0 -1
  199. package/dist/concurrency-MUjT7VjM.js +0 -109
  200. package/dist/concurrency-MUjT7VjM.js.map +0 -1
  201. package/dist/cost-ledger-DHAjwNj7.js.map +0 -1
  202. package/dist/default-registry-Brxr728w.d.ts.map +0 -1
  203. package/dist/default-registry-CHmdy2An.js.map +0 -1
  204. package/dist/errors-8YnH8WlF.js.map +0 -1
  205. package/dist/errors-CEk209JS.d.ts.map +0 -1
  206. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +0 -1
  207. package/dist/index-B4Fjfo5U.d.ts.map +0 -1
  208. package/dist/index-CyC1BTmn.d.ts.map +0 -1
  209. package/dist/index-DuhJaaiH.d.ts.map +0 -1
  210. package/dist/ledger-core-DAKFKRzi.js.map +0 -1
  211. package/dist/llm-client-BiK4HW0u.d.ts +0 -290
  212. package/dist/llm-client-BiK4HW0u.d.ts.map +0 -1
  213. package/dist/llm-client-DHx8pzyJ.js.map +0 -1
  214. package/dist/proposal-findings-DCawte-y.js.map +0 -1
  215. package/dist/raw-provider-sink-BU29Sh8h.d.ts +0 -134
  216. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +0 -1
  217. package/dist/replay-BRfMIs81.d.ts.map +0 -1
  218. package/dist/replay-C6wRg47C.js.map +0 -1
  219. package/dist/researcher-Doo95b50.d.ts.map +0 -1
  220. package/dist/semantic-concept-judge-Btozx3Vc.js.map +0 -1
  221. package/dist/skill-usage-BDQVPIG1.d.ts.map +0 -1
  222. package/dist/skillopt-optimization-method-0UmPD6aP.js.map +0 -1
  223. package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +0 -1
  224. package/dist/store-CxJry_cs.d.ts +0 -229
  225. package/dist/store-CxJry_cs.d.ts.map +0 -1
  226. package/dist/supervisor-run-Dr5HnTup.js.map +0 -1
  227. package/dist/tools-D8yTtNSN.js +0 -1190
  228. package/dist/tools-D8yTtNSN.js.map +0 -1
  229. package/dist/types-Cc3qbqzj.d.ts +0 -387
  230. package/dist/types-Cc3qbqzj.d.ts.map +0 -1
  231. package/dist/types-DVjczBM9.d.ts.map +0 -1
@@ -0,0 +1,554 @@
1
+ import { l as assertValidAnalystUsageReceipt } from "./single-run-lock-D_bS5xhj.js";
2
+ import { performance } from "node:perf_hooks";
3
+ import { linearSumAssignment } from "linear-sum-assignment";
4
+ //#region src/analyst/benchmark-scoring.ts
5
+ function scoreAnalystFindings(testCase, findings) {
6
+ assertValidAnalystScoringCase(testCase);
7
+ const matchedFindingByIssue = matchFindingsToIssues(testCase.expectedIssues, findings);
8
+ const matchedIssueIds = testCase.expectedIssues.filter((_, index) => matchedFindingByIssue.has(index)).map((issue) => issue.id);
9
+ const missedIssueIds = testCase.expectedIssues.filter((_, index) => !matchedFindingByIssue.has(index)).map((issue) => issue.id);
10
+ const supportedFindingIndexes = new Set(matchedFindingByIssue.values());
11
+ const unsupportedFindingIndexes = findings.map((_, index) => index).filter((index) => !supportedFindingIndexes.has(index));
12
+ const expectedIssueCount = testCase.expectedIssues.length;
13
+ const issueRecall = expectedIssueCount === 0 ? 1 : matchedIssueIds.length / expectedIssueCount;
14
+ const findingPrecision = findings.length === 0 ? expectedIssueCount === 0 ? 1 : 0 : supportedFindingIndexes.size / findings.length;
15
+ const f1 = harmonicMeanScore(findingPrecision, issueRecall);
16
+ const allEvidence = findings.flatMap((finding) => finding.evidence_refs);
17
+ const criticalIssues = testCase.expectedIssues.filter((issue) => (issue.criticalEvidence?.length ?? 0) > 0);
18
+ const criticalHits = testCase.expectedIssues.filter((issue) => {
19
+ if ((issue.criticalEvidence?.length ?? 0) === 0) return false;
20
+ return matchesEvidence(allEvidence, issue.criticalEvidence ?? [], "any");
21
+ }).length;
22
+ const findingsWithEvidence = findings.filter((finding) => finding.evidence_refs.length > 0).length;
23
+ const unlabeledEvidence = testCase.labeledEvidence ? allEvidence.filter((ref) => !testCase.labeledEvidence.some((expected) => evidenceMatches(ref, expected))) : [];
24
+ return {
25
+ expectedIssueCount,
26
+ matchedIssueIds,
27
+ missedIssueIds,
28
+ supportedFindingIndexes: [...supportedFindingIndexes].sort((a, b) => a - b),
29
+ unsupportedFindingIndexes,
30
+ unlabeledEvidence,
31
+ issueRecall,
32
+ findingPrecision,
33
+ f1,
34
+ criticalStepAccuracy: criticalIssues.length === 0 ? null : criticalHits / criticalIssues.length,
35
+ citationCoverage: findings.length === 0 ? null : findingsWithEvidence / findings.length,
36
+ citationExcerptCoverage: allEvidence.length === 0 ? null : allEvidence.filter((evidence) => Boolean(evidence.excerpt?.trim())).length / allEvidence.length,
37
+ citationLabelAgreement: testCase.labeledEvidence === void 0 ? null : allEvidence.length === 0 ? findings.length === 0 ? null : 0 : (allEvidence.length - unlabeledEvidence.length) / allEvidence.length,
38
+ predictionOnLabelEmptyCase: expectedIssueCount === 0 && findings.length > 0
39
+ };
40
+ }
41
+ function assertValidAnalystScoringCase(testCase) {
42
+ if (!testCase.id.trim()) throw new TypeError("analyst benchmark case id must not be empty");
43
+ const ids = /* @__PURE__ */ new Set();
44
+ for (const issue of testCase.expectedIssues) {
45
+ if (!issue.id.trim()) throw new TypeError(`${testCase.id}: expected issue id must not be empty`);
46
+ if (ids.has(issue.id)) throw new TypeError(`${testCase.id}: duplicate expected issue id '${issue.id}'`);
47
+ ids.add(issue.id);
48
+ if (!issue.findingIds?.length && !issue.areas?.length && !issue.subjects?.length && !issue.evidence?.length) throw new TypeError(`${testCase.id}/${issue.id}: expected issue must identify a finding by id, area, subject, or evidence`);
49
+ }
50
+ for (const ref of testCase.labeledEvidence ?? []) if (!ref.uri.trim()) throw new TypeError(`${testCase.id}: labeled evidence URI must not be empty`);
51
+ }
52
+ function harmonicMeanScore(a, b) {
53
+ return a + b === 0 ? 0 : 2 * a * b / (a + b);
54
+ }
55
+ function matchFindingsToIssues(issues, findings) {
56
+ if (issues.length === 0) return /* @__PURE__ */ new Map();
57
+ const cardinalityWeight = issues.length + 1;
58
+ const assignment = linearSumAssignment(issues.map((issue) => [...findings.map((finding) => {
59
+ if (!findingMatchesIssue(finding, issue)) return -1;
60
+ const criticalHit = (issue.criticalEvidence?.length ?? 0) > 0 && matchesEvidence(finding.evidence_refs, issue.criticalEvidence ?? [], "any");
61
+ return cardinalityWeight + Number(criticalHit);
62
+ }), ...Array.from({ length: issues.length }, () => 0)]), { maximaze: true }).rowAssignments;
63
+ const matches = /* @__PURE__ */ new Map();
64
+ for (const [issueIndex, column] of assignment.entries()) {
65
+ if (column < 0 || column >= findings.length) continue;
66
+ if (!findingMatchesIssue(findings[column], issues[issueIndex])) continue;
67
+ matches.set(issueIndex, column);
68
+ }
69
+ return matches;
70
+ }
71
+ function findingMatchesIssue(finding, issue) {
72
+ if (issue.findingIds && !issue.findingIds.includes(finding.finding_id)) return false;
73
+ if (issue.areas && !issue.areas.includes(finding.area)) return false;
74
+ if (issue.subjects && (!finding.subject || !issue.subjects.includes(finding.subject))) return false;
75
+ if (issue.evidence && !matchesEvidence(finding.evidence_refs, issue.evidence, issue.evidenceMode ?? "any")) return false;
76
+ return true;
77
+ }
78
+ function matchesEvidence(actual, expected, mode) {
79
+ if (expected.length === 0) return true;
80
+ const match = (target) => actual.some((ref) => evidenceMatches(ref, target));
81
+ return mode === "all" ? expected.every(match) : expected.some(match);
82
+ }
83
+ function evidenceMatches(actual, expected) {
84
+ return actual.uri === expected.uri && (expected.kind === void 0 || actual.kind === expected.kind);
85
+ }
86
+ //#endregion
87
+ //#region src/analyst/benchmark-summary.ts
88
+ function summarizeAnalystBenchmarkRunner(runnerId, observations) {
89
+ const issueBearing = observations.filter((observation) => observation.labelState === "positive");
90
+ const completed = observations.filter((observation) => !observation.error);
91
+ const expectedIssues = issueBearing.reduce((sum, observation) => sum + observation.score.expectedIssueCount, 0);
92
+ const matchedIssues = issueBearing.reduce((sum, observation) => sum + observation.score.matchedIssueIds.length, 0);
93
+ const issueFindings = issueBearing.reduce((sum, observation) => sum + (observation.error ? 0 : observation.findings.length), 0);
94
+ const supportedFindings = issueBearing.reduce((sum, observation) => sum + observation.score.supportedFindingIndexes.length, 0);
95
+ const issueRecall = expectedIssues === 0 ? null : matchedIssues / expectedIssues;
96
+ const findingPrecision = expectedIssues === 0 ? null : issueFindings === 0 ? 0 : supportedFindings / issueFindings;
97
+ const macroIssueRecall = issueBearing.length === 0 ? null : mean(issueBearing.map((observation) => observation.score.issueRecall));
98
+ const macroFindingPrecision = issueBearing.length === 0 ? null : mean(issueBearing.map((observation) => observation.score.findingPrecision));
99
+ const macroF1 = issueBearing.length === 0 ? null : mean(issueBearing.map((observation) => observation.score.f1));
100
+ const critical = observations.map((observation) => observation.score.criticalStepAccuracy).filter((value) => value !== null);
101
+ const findingsWithEvidence = completed.reduce((sum, observation) => sum + observation.findings.filter((finding) => finding.evidence_refs.length > 0).length, 0);
102
+ const allFindings = completed.reduce((sum, observation) => sum + observation.findings.length, 0);
103
+ const allCitations = completed.flatMap((observation) => observation.findings.flatMap((finding) => finding.evidence_refs));
104
+ const citationObservations = completed.filter((observation) => observation.score.citationLabelAgreement !== null);
105
+ const citationCount = citationObservations.reduce((sum, observation) => sum + observation.findings.reduce((count, finding) => count + finding.evidence_refs.length, 0), 0);
106
+ const invalidCitationCount = citationObservations.reduce((sum, observation) => sum + observation.score.unlabeledEvidence.length, 0);
107
+ const trustedNegative = observations.filter((observation) => observation.labelState === "trusted-negative");
108
+ const completedTrustedNegative = trustedNegative.filter((observation) => !observation.error);
109
+ const unlabeled = observations.filter((observation) => observation.labelState === "unlabeled");
110
+ const completedUnlabeled = unlabeled.filter((observation) => !observation.error);
111
+ const resolvedCitations = completed.reduce((sum, observation) => sum + (observation.evidenceResolution?.resolved ?? 0), 0);
112
+ const unresolvedCitations = completed.reduce((sum, observation) => sum + (observation.evidenceResolution?.unresolvedEvidence.length ?? 0), 0);
113
+ const citationResolutionErrors = completed.reduce((sum, observation) => sum + (observation.evidenceResolution?.errors.length ?? 0), 0);
114
+ const resolutionAttempts = completed.filter((observation) => observation.findings.some((finding) => finding.evidence_refs.length > 0));
115
+ const citationResolutionUnknownRuns = resolutionAttempts.filter((observation) => !observation.evidenceResolution || observation.evidenceResolution.errors.length > 0).length;
116
+ const usages = observations.map((observation) => observation.usage);
117
+ const knownCostUsd = stableSum(usages.map((usage) => {
118
+ if (!usage) return 0;
119
+ return usage.cost.kind === "uncaptured" ? usage.knownCostUsd ?? 0 : usage.cost.usd;
120
+ }));
121
+ const predictionAgreement = repeatedObservationAgreement(observations, predictionSignature);
122
+ const matchedLabelAgreement = repeatedObservationAgreement(observations.filter((observation) => observation.labelState === "positive"), matchedLabelSignature);
123
+ return {
124
+ runnerId,
125
+ plannedRuns: observations.length,
126
+ completedRuns: observations.filter((observation) => !observation.error).length,
127
+ failedRuns: observations.filter((observation) => Boolean(observation.error)).length,
128
+ issueBearingRuns: issueBearing.length,
129
+ trustedNegativeRuns: trustedNegative.length,
130
+ unlabeledRuns: unlabeled.length,
131
+ issueRecall,
132
+ findingPrecision,
133
+ f1: findingPrecision === null || issueRecall === null ? null : harmonicMeanScore(findingPrecision, issueRecall),
134
+ macroIssueRecall,
135
+ macroFindingPrecision,
136
+ macroF1,
137
+ criticalStepAccuracy: critical.length === 0 ? null : mean(critical),
138
+ citationCoverage: allFindings === 0 ? null : findingsWithEvidence / allFindings,
139
+ citationExcerptCoverage: allCitations.length === 0 ? null : allCitations.filter((evidence) => Boolean(evidence.excerpt?.trim())).length / allCitations.length,
140
+ citationLabelAgreement: citationObservations.length === 0 ? null : citationCount === 0 ? 0 : (citationCount - invalidCitationCount) / citationCount,
141
+ citationResolution: resolutionAttempts.length === 0 || citationResolutionUnknownRuns > 0 || resolvedCitations + unresolvedCitations === 0 ? null : resolvedCitations / (resolvedCitations + unresolvedCitations),
142
+ citationResolutionUnknownRuns,
143
+ unresolvedCitations,
144
+ citationResolutionErrors,
145
+ trustedNegativeFalsePositiveRate: completedTrustedNegative.length === 0 ? null : completedTrustedNegative.filter((observation) => observation.score.predictionOnLabelEmptyCase).length / completedTrustedNegative.length,
146
+ trustedNegativeFailureRate: trustedNegative.length === 0 ? null : trustedNegative.filter((observation) => Boolean(observation.error)).length / trustedNegative.length,
147
+ unlabeledPredictionRate: completedUnlabeled.length === 0 ? null : completedUnlabeled.filter((observation) => observation.findings.length > 0).length / completedUnlabeled.length,
148
+ unlabeledFailureRate: unlabeled.length === 0 ? null : unlabeled.filter((observation) => Boolean(observation.error)).length / unlabeled.length,
149
+ predictionAgreement: predictionAgreement.value,
150
+ predictionAgreementCases: predictionAgreement.cases,
151
+ matchedLabelAgreement: matchedLabelAgreement.value,
152
+ matchedLabelAgreementCases: matchedLabelAgreement.cases,
153
+ latencyMs: latencyDistribution(observations.map((observation) => observation.latencyMs).filter((value) => value !== null)),
154
+ benchmarkClockLatencyRuns: observations.filter((observation) => observation.latencySource === "benchmark-clock").length,
155
+ runnerReportedLatencyRuns: observations.filter((observation) => observation.latencySource === "runner-reported").length,
156
+ latencyUnknownRuns: observations.filter((observation) => observation.latencySource === "uncaptured").length,
157
+ calls: usages.reduce((sum, usage) => sum + (usage?.calls ?? 0), 0),
158
+ callsUnknownRuns: usages.filter((usage) => !usage || usage.calls === null).length,
159
+ inputTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.input ?? 0), 0),
160
+ outputTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.output ?? 0), 0),
161
+ reasoningTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.reasoning ?? 0), 0),
162
+ cachedTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.cached ?? 0), 0),
163
+ cacheWriteTokens: usages.reduce((sum, usage) => sum + (usage?.tokens?.cacheWrite ?? 0), 0),
164
+ tokenUsageUnknownRuns: usages.filter((usage) => !usage?.tokens).length,
165
+ reasoningTokenUsageUnknownRuns: usages.filter((usage) => usage?.tokens?.reasoning === void 0).length,
166
+ cachedTokenUsageUnknownRuns: usages.filter((usage) => usage?.tokens?.cached === void 0).length,
167
+ cacheWriteTokenUsageUnknownRuns: usages.filter((usage) => usage?.tokens?.cacheWrite === void 0).length,
168
+ knownCostUsd,
169
+ costUnknownRuns: usages.filter((usage) => !usage || usage.cost.kind === "uncaptured").length
170
+ };
171
+ }
172
+ function mean(values) {
173
+ return values.length === 0 ? 0 : stableSum(values) / values.length;
174
+ }
175
+ function stableSum(values) {
176
+ const ordered = [...values].sort((left, right) => Math.abs(left) - Math.abs(right) || left - right);
177
+ let sum = 0;
178
+ let correction = 0;
179
+ for (const value of ordered) {
180
+ const next = sum + value;
181
+ correction += Math.abs(sum) >= Math.abs(value) ? sum - next + value : value - next + sum;
182
+ sum = next;
183
+ }
184
+ return sum + correction;
185
+ }
186
+ function latencyDistribution(values) {
187
+ if (values.length === 0) return null;
188
+ const sorted = [...values].sort((a, b) => a - b);
189
+ return {
190
+ min: sorted[0],
191
+ mean: mean(sorted),
192
+ p50: percentile(sorted, .5),
193
+ p95: percentile(sorted, .95),
194
+ max: sorted.at(-1)
195
+ };
196
+ }
197
+ function percentile(sorted, quantile) {
198
+ if (sorted.length === 0) return 0;
199
+ return sorted[Math.ceil(quantile * sorted.length) - 1] ?? sorted.at(-1) ?? 0;
200
+ }
201
+ function repeatedObservationAgreement(observations, signature) {
202
+ const byCase = /* @__PURE__ */ new Map();
203
+ for (const observation of observations) {
204
+ const rows = byCase.get(observation.caseId) ?? [];
205
+ rows.push(observation);
206
+ byCase.set(observation.caseId, rows);
207
+ }
208
+ const caseAgreements = [];
209
+ for (const rows of byCase.values()) {
210
+ const agreements = [];
211
+ for (let left = 0; left < rows.length; left++) for (let right = left + 1; right < rows.length; right++) agreements.push(jaccard(signature(rows[left]), signature(rows[right])));
212
+ if (agreements.length > 0) caseAgreements.push(mean(agreements));
213
+ }
214
+ return {
215
+ value: caseAgreements.length === 0 ? null : mean(caseAgreements),
216
+ cases: caseAgreements.length
217
+ };
218
+ }
219
+ function matchedLabelSignature(observation) {
220
+ if (observation.error) return [`error:${observation.error.class}`];
221
+ return observation.score.matchedIssueIds;
222
+ }
223
+ function predictionSignature(observation) {
224
+ if (observation.error) return [`error:${observation.error.class}`];
225
+ return observation.findings.map((finding) => JSON.stringify([finding.finding_id, finding.evidence_refs.map((evidence) => [
226
+ evidence.kind,
227
+ evidence.uri,
228
+ evidence.excerpt ?? null
229
+ ]).sort((left, right) => JSON.stringify(left).localeCompare(JSON.stringify(right)))])).sort();
230
+ }
231
+ function jaccard(left, right) {
232
+ const a = new Set(left);
233
+ const b = new Set(right);
234
+ const union = /* @__PURE__ */ new Set([...a, ...b]);
235
+ if (union.size === 0) return 1;
236
+ let intersection = 0;
237
+ for (const value of a) if (b.has(value)) intersection += 1;
238
+ return intersection / union.size;
239
+ }
240
+ //#endregion
241
+ //#region src/analyst/benchmark.ts
242
+ /**
243
+ * Resolve canonical `trace://<trace>/span/<span>` evidence against a trace store.
244
+ * Other evidence kinds and URI schemes require a caller-supplied resolver.
245
+ */
246
+ function traceStoreEvidenceResolver(getStore) {
247
+ return async ({ caseInput, evidence, signal }) => {
248
+ if (evidence.kind !== "span") return false;
249
+ const location = parseTraceSpanUri(evidence.uri);
250
+ if (!location) return false;
251
+ const result = await getStore(caseInput).viewSpans({
252
+ trace_id: location.traceId,
253
+ span_ids: [location.spanId]
254
+ }, signal ? { signal } : void 0);
255
+ return result.trace_id === location.traceId && result.missing_span_ids.length === 0 && result.spans.some((span) => span.span_id === location.spanId);
256
+ };
257
+ }
258
+ async function runAnalystBenchmark(options) {
259
+ validateBenchmarkOptions(options);
260
+ const startedAt = (/* @__PURE__ */ new Date()).toISOString();
261
+ const repetitions = options.repetitions ?? 1;
262
+ const runnerOrderSeed = options.runnerOrderSeed ?? 0;
263
+ const allJobs = benchmarkJobs(options.cases, options.runners, repetitions, runnerOrderSeed);
264
+ const maxConcurrency = Math.min(options.maxConcurrency ?? 1, allJobs.length);
265
+ const initialObservations = validateInitialObservations(options.initialObservations ?? [], allJobs);
266
+ const completed = new Set(initialObservations.map(observationKey));
267
+ const jobs = allJobs.filter((job) => !completed.has(jobKey(job)));
268
+ const observations = [...initialObservations];
269
+ let cursor = 0;
270
+ const worker = async () => {
271
+ while (cursor < jobs.length) {
272
+ options.signal?.throwIfAborted();
273
+ const job = jobs[cursor++];
274
+ const observation = await runBenchmarkJob(job, options.signal, options.resolveEvidence);
275
+ options.signal?.throwIfAborted();
276
+ await options.onObservation?.(observation);
277
+ options.signal?.throwIfAborted();
278
+ observations.push(observation);
279
+ }
280
+ };
281
+ await Promise.all(Array.from({ length: Math.min(maxConcurrency, jobs.length) }, worker));
282
+ options.signal?.throwIfAborted();
283
+ const runnerOrder = new Map(options.runners.map((runner, index) => [runner.id, index]));
284
+ const caseOrder = new Map(options.cases.map((testCase, index) => [testCase.id, index]));
285
+ observations.sort((a, b) => (runnerOrder.get(a.runnerId) ?? 0) - (runnerOrder.get(b.runnerId) ?? 0) || (caseOrder.get(a.caseId) ?? 0) - (caseOrder.get(b.caseId) ?? 0) || a.repetition - b.repetition);
286
+ return {
287
+ provenance: {
288
+ ...options.benchmark,
289
+ startedAt,
290
+ endedAt: (/* @__PURE__ */ new Date()).toISOString(),
291
+ caseCount: options.cases.length,
292
+ runnerIds: options.runners.map((runner) => runner.id),
293
+ repetitions,
294
+ maxConcurrency,
295
+ runnerOrderSeed
296
+ },
297
+ observations,
298
+ summaries: options.runners.map((runner) => summarizeAnalystBenchmarkRunner(runner.id, observations.filter((observation) => observation.runnerId === runner.id)))
299
+ };
300
+ }
301
+ function registryBenchmarkRunner(options) {
302
+ return {
303
+ id: options.id,
304
+ async analyze(input, context) {
305
+ const result = await options.registry.run(`${options.id}:${context.caseId}:${context.repetition}`, input, {
306
+ ...options.runOptions,
307
+ signal: context.signal
308
+ });
309
+ return {
310
+ findings: result.findings,
311
+ usage: mergeRegistryUsage(result),
312
+ metadata: { analystRun: result },
313
+ ...options.failOnAnalystFailure ? { error: registryRunFailure(result) } : {}
314
+ };
315
+ }
316
+ };
317
+ }
318
+ async function runBenchmarkJob(job, signal, resolveEvidence) {
319
+ const started = performance.now();
320
+ try {
321
+ const output = await job.runner.analyze(job.testCase.input, {
322
+ caseId: job.testCase.id,
323
+ repetition: job.repetition,
324
+ signal
325
+ });
326
+ if (output.usage) assertValidAnalystUsageReceipt(output.usage, "analyst benchmark usage");
327
+ const benchmarkLatencyMs = performance.now() - started;
328
+ const latency = resolveBenchmarkLatency(output.observedLatencyMs, benchmarkLatencyMs);
329
+ const scoredFindings = output.error ? [] : output.findings;
330
+ return {
331
+ runnerId: job.runner.id,
332
+ caseId: job.testCase.id,
333
+ clusterId: job.testCase.clusterId,
334
+ labelState: job.testCase.labelState,
335
+ repetition: job.repetition,
336
+ executionIndex: job.executionIndex,
337
+ latencyMs: latency.value,
338
+ latencySource: latency.source,
339
+ findings: output.findings,
340
+ score: scoreAnalystFindings(job.testCase, scoredFindings),
341
+ evidenceResolution: resolveEvidence ? await resolveFindingEvidence(job.testCase, output.findings, resolveEvidence, signal) : void 0,
342
+ caseTags: [...job.testCase.tags ?? []],
343
+ caseMetadata: job.testCase.metadata,
344
+ usage: output.usage,
345
+ runnerMetadata: output.metadata,
346
+ ...output.error ? { error: output.error } : {}
347
+ };
348
+ } catch (error) {
349
+ if (signal?.aborted) throw error;
350
+ const findings = [];
351
+ return {
352
+ runnerId: job.runner.id,
353
+ caseId: job.testCase.id,
354
+ clusterId: job.testCase.clusterId,
355
+ labelState: job.testCase.labelState,
356
+ repetition: job.repetition,
357
+ executionIndex: job.executionIndex,
358
+ latencyMs: performance.now() - started,
359
+ latencySource: "benchmark-clock",
360
+ findings,
361
+ score: scoreAnalystFindings(job.testCase, findings),
362
+ caseTags: [...job.testCase.tags ?? []],
363
+ caseMetadata: job.testCase.metadata,
364
+ error: {
365
+ class: error instanceof Error ? error.constructor.name : "Error",
366
+ message: error instanceof Error ? error.message : String(error)
367
+ }
368
+ };
369
+ }
370
+ }
371
+ function resolveBenchmarkLatency(observedLatencyMs, fallbackMs) {
372
+ if (observedLatencyMs === void 0) return {
373
+ value: fallbackMs,
374
+ source: "benchmark-clock"
375
+ };
376
+ if (observedLatencyMs === null) return {
377
+ value: null,
378
+ source: "uncaptured"
379
+ };
380
+ if (!Number.isFinite(observedLatencyMs) || observedLatencyMs < 0) throw new RangeError("analyst benchmark observedLatencyMs must be finite and non-negative");
381
+ return {
382
+ value: observedLatencyMs,
383
+ source: "runner-reported"
384
+ };
385
+ }
386
+ function registryRunFailure(result) {
387
+ const failed = result.per_analyst.filter((summary) => summary.status === "failed");
388
+ if (failed.length === 0) return void 0;
389
+ return {
390
+ class: "AnalystRunFailure",
391
+ message: failed.map((summary) => `${summary.analyst_id}: ${summary.error?.class ?? "Error"}: ${summary.error?.message ?? "analyst failed"}`).join("; ")
392
+ };
393
+ }
394
+ function benchmarkJobs(cases, runners, repetitions, runnerOrderSeed) {
395
+ const seededRunners = [...runners].sort((left, right) => stableHash(`${runnerOrderSeed}\u0000${left.id}`) - stableHash(`${runnerOrderSeed}\u0000${right.id}`) || left.id.localeCompare(right.id));
396
+ let executionIndex = 0;
397
+ return cases.flatMap((testCase, caseIndex) => Array.from({ length: repetitions }, (_, repetition) => {
398
+ const rotation = (caseIndex * repetitions + repetition) % seededRunners.length;
399
+ return [...seededRunners.slice(rotation), ...seededRunners.slice(0, rotation)].map((runner) => ({
400
+ runner,
401
+ testCase,
402
+ repetition,
403
+ executionIndex: executionIndex++
404
+ }));
405
+ }).flat());
406
+ }
407
+ function validateInitialObservations(observations, jobs) {
408
+ const expected = new Map(jobs.map((job) => [jobKey(job), job]));
409
+ const seen = /* @__PURE__ */ new Set();
410
+ return observations.map((observation) => {
411
+ const key = observationKey(observation);
412
+ if (seen.has(key)) throw new TypeError(`duplicate initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}'`);
413
+ seen.add(key);
414
+ const job = expected.get(key);
415
+ if (!job) throw new TypeError(`initial analyst benchmark observation does not match a planned job: '${observation.runnerId}/${observation.caseId}/${observation.repetition}'`);
416
+ if (observation.executionIndex !== job.executionIndex) throw new TypeError(`initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' has executionIndex ${observation.executionIndex}; expected ${job.executionIndex}`);
417
+ if (observation.clusterId !== job.testCase.clusterId || observation.labelState !== job.testCase.labelState) throw new TypeError(`initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' does not match the current case labels`);
418
+ if (JSON.stringify(observation.caseTags) !== JSON.stringify(job.testCase.tags ?? []) || JSON.stringify(observation.caseMetadata) !== JSON.stringify(job.testCase.metadata)) throw new TypeError(`initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' does not match the current case metadata`);
419
+ const expectedScore = scoreAnalystFindings(job.testCase, observation.error ? [] : observation.findings);
420
+ if (JSON.stringify(observation.score) !== JSON.stringify(expectedScore)) throw new TypeError(`initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' has stale or invalid scores`);
421
+ if (observation.usage) assertValidAnalystUsageReceipt(observation.usage, `initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' usage`);
422
+ if (![
423
+ "benchmark-clock",
424
+ "runner-reported",
425
+ "uncaptured"
426
+ ].includes(observation.latencySource) || observation.latencySource === "uncaptured" && observation.latencyMs !== null || observation.latencySource !== "uncaptured" && (observation.latencyMs === null || !Number.isFinite(observation.latencyMs) || observation.latencyMs < 0)) throw new TypeError(`initial analyst benchmark observation '${observation.runnerId}/${observation.caseId}/${observation.repetition}' has invalid latency`);
427
+ return { ...observation };
428
+ });
429
+ }
430
+ function jobKey(job) {
431
+ return `${job.runner.id}\u0000${job.testCase.id}\u0000${job.repetition}`;
432
+ }
433
+ function observationKey(observation) {
434
+ return `${observation.runnerId}\u0000${observation.caseId}\u0000${observation.repetition}`;
435
+ }
436
+ async function resolveFindingEvidence(testCase, findings, resolver, signal) {
437
+ const evidence = findings.flatMap((finding) => finding.evidence_refs);
438
+ const resolved = [];
439
+ const unresolvedEvidence = [];
440
+ const errors = [];
441
+ for (const ref of evidence) {
442
+ signal?.throwIfAborted();
443
+ try {
444
+ if (await resolver({
445
+ caseId: testCase.id,
446
+ caseInput: testCase.input,
447
+ evidence: ref,
448
+ signal
449
+ })) resolved.push(ref);
450
+ else unresolvedEvidence.push(ref);
451
+ } catch (error) {
452
+ if (signal?.aborted) throw error;
453
+ errors.push({
454
+ evidence: ref,
455
+ class: error instanceof Error ? error.constructor.name : "Error",
456
+ message: error instanceof Error ? error.message : String(error)
457
+ });
458
+ }
459
+ }
460
+ return {
461
+ checked: evidence.length,
462
+ resolved: resolved.length,
463
+ unresolvedEvidence,
464
+ errors,
465
+ validity: evidence.length === 0 || errors.length > 0 ? null : resolved.length / evidence.length
466
+ };
467
+ }
468
+ function parseTraceSpanUri(uri) {
469
+ const match = /^trace:\/\/([^/]+)\/span\/([^/]+)$/.exec(uri);
470
+ if (!match) return null;
471
+ try {
472
+ const traceId = decodeURIComponent(match[1]);
473
+ const spanId = decodeURIComponent(match[2]);
474
+ return traceId && spanId ? {
475
+ traceId,
476
+ spanId
477
+ } : null;
478
+ } catch {
479
+ return null;
480
+ }
481
+ }
482
+ function validateBenchmarkOptions(options) {
483
+ if (options.cases.length === 0) throw new TypeError("runAnalystBenchmark requires cases");
484
+ if (options.runners.length === 0) throw new TypeError("runAnalystBenchmark requires runners");
485
+ const repetitions = options.repetitions ?? 1;
486
+ const maxConcurrency = options.maxConcurrency ?? 1;
487
+ if (!Number.isSafeInteger(repetitions) || repetitions < 1) throw new RangeError("runAnalystBenchmark repetitions must be a positive safe integer");
488
+ if (!Number.isSafeInteger(maxConcurrency) || maxConcurrency < 1) throw new RangeError("runAnalystBenchmark maxConcurrency must be a positive safe integer");
489
+ if (!Number.isSafeInteger(options.runnerOrderSeed ?? 0)) throw new RangeError("runAnalystBenchmark runnerOrderSeed must be a safe integer");
490
+ assertUniqueNonEmpty(options.cases.map((testCase) => testCase.id), "case");
491
+ assertUniqueNonEmpty(options.runners.map((runner) => runner.id), "runner");
492
+ for (const testCase of options.cases) validateBenchmarkCase(testCase);
493
+ }
494
+ function validateBenchmarkCase(testCase) {
495
+ assertValidAnalystScoringCase(testCase);
496
+ if (!testCase.clusterId.trim()) throw new TypeError(`${testCase.id}: analyst benchmark clusterId must not be empty`);
497
+ if (testCase.labelState !== "positive" && testCase.labelState !== "trusted-negative" && testCase.labelState !== "unlabeled") throw new TypeError(`${testCase.id}: analyst benchmark labelState is invalid`);
498
+ if (testCase.labelState === "positive" && testCase.expectedIssues.length === 0) throw new TypeError(`${testCase.id}: positive case requires at least one expected issue`);
499
+ if (testCase.labelState !== "positive" && testCase.expectedIssues.length > 0) throw new TypeError(`${testCase.id}: ${testCase.labelState} case cannot contain expected issues`);
500
+ }
501
+ function assertUniqueNonEmpty(values, label) {
502
+ const seen = /* @__PURE__ */ new Set();
503
+ for (const value of values) {
504
+ if (!value.trim()) throw new TypeError(`analyst benchmark ${label} id must not be empty`);
505
+ if (seen.has(value)) throw new TypeError(`duplicate analyst benchmark ${label} id '${value}'`);
506
+ seen.add(value);
507
+ }
508
+ }
509
+ function stableHash(value) {
510
+ let hash = 2166136261;
511
+ for (let index = 0; index < value.length; index += 1) {
512
+ hash ^= value.charCodeAt(index);
513
+ hash = Math.imul(hash, 16777619);
514
+ }
515
+ return hash >>> 0;
516
+ }
517
+ function mergeRegistryUsage(result) {
518
+ const usages = result.per_analyst.map((summary) => summary.usage);
519
+ const calls = usages.every((usage) => usage.calls !== null) ? usages.reduce((sum, usage) => sum + (usage.calls ?? 0), 0) : null;
520
+ const tokens = usages.every((usage) => usage.tokens !== null) ? usages.reduce((sum, usage) => ({
521
+ input: sum.input + (usage.tokens?.input ?? 0),
522
+ output: sum.output + (usage.tokens?.output ?? 0),
523
+ reasoning: sum.reasoning + (usage.tokens?.reasoning ?? 0),
524
+ cached: sum.cached + (usage.tokens?.cached ?? 0),
525
+ cacheWrite: sum.cacheWrite + (usage.tokens?.cacheWrite ?? 0)
526
+ }), {
527
+ input: 0,
528
+ output: 0,
529
+ reasoning: 0,
530
+ cached: 0,
531
+ cacheWrite: 0
532
+ }) : null;
533
+ const knownCostUsd = usages.reduce((sum, usage) => sum + (usage.cost.kind === "uncaptured" ? usage.knownCostUsd ?? 0 : usage.cost.usd), 0);
534
+ const cost = usages.some((usage) => usage.cost.kind === "uncaptured") ? {
535
+ kind: "uncaptured",
536
+ usd: null
537
+ } : usages.some((usage) => usage.cost.kind === "estimated") ? {
538
+ kind: "estimated",
539
+ usd: knownCostUsd
540
+ } : {
541
+ kind: "observed",
542
+ usd: knownCostUsd
543
+ };
544
+ return {
545
+ calls,
546
+ tokens,
547
+ cost,
548
+ ...cost.kind === "uncaptured" ? { knownCostUsd } : {}
549
+ };
550
+ }
551
+ //#endregion
552
+ export { scoreAnalystFindings as a, summarizeAnalystBenchmarkRunner as i, runAnalystBenchmark as n, traceStoreEvidenceResolver as r, registryBenchmarkRunner as t };
553
+
554
+ //# sourceMappingURL=benchmark-D8dkki-J.js.map