@tangle-network/agent-eval 0.129.0 → 0.130.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (427) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +1 -0
  3. package/dist/active-curriculum-C4mk67HP.js +214 -0
  4. package/dist/active-curriculum-C4mk67HP.js.map +1 -0
  5. package/dist/adversarial-smnADNFS.d.ts +21 -0
  6. package/dist/adversarial-smnADNFS.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +81 -2872
  8. package/dist/analyst/index.d.ts.map +1 -0
  9. package/dist/analyst/index.js +319 -360
  10. package/dist/analyst/index.js.map +1 -1
  11. package/dist/analyst-BkTS3C58.d.ts +89 -0
  12. package/dist/analyst-BkTS3C58.d.ts.map +1 -0
  13. package/dist/analyst-LsnNpSkm.js +152 -0
  14. package/dist/analyst-LsnNpSkm.js.map +1 -0
  15. package/dist/analyze-runs-C1CavBMk.js +1067 -0
  16. package/dist/analyze-runs-C1CavBMk.js.map +1 -0
  17. package/dist/analyze-runs-FsgYCinh.d.ts +72 -0
  18. package/dist/analyze-runs-FsgYCinh.d.ts.map +1 -0
  19. package/dist/attribute-vocabulary-DLJ6303h.d.ts +54 -0
  20. package/dist/attribute-vocabulary-DLJ6303h.d.ts.map +1 -0
  21. package/dist/authenticity/index.d.ts +81 -79
  22. package/dist/authenticity/index.d.ts.map +1 -0
  23. package/dist/authenticity/index.js +209 -193
  24. package/dist/authenticity/index.js.map +1 -1
  25. package/dist/baseline-HsBvw_dk.js +550 -0
  26. package/dist/baseline-HsBvw_dk.js.map +1 -0
  27. package/dist/baseline-hG3K85h4.d.ts +125 -0
  28. package/dist/baseline-hG3K85h4.d.ts.map +1 -0
  29. package/dist/belief-state/index.d.ts +448 -1188
  30. package/dist/belief-state/index.d.ts.map +1 -0
  31. package/dist/belief-state/index.js +1617 -1709
  32. package/dist/belief-state/index.js.map +1 -1
  33. package/dist/benchmarks/index.d.ts +2 -891
  34. package/dist/benchmarks/index.js +2 -60
  35. package/dist/benchmarks-DviOvUNr.js +754 -0
  36. package/dist/benchmarks-DviOvUNr.js.map +1 -0
  37. package/dist/builder-eval/index.d.ts +150 -662
  38. package/dist/builder-eval/index.d.ts.map +1 -0
  39. package/dist/builder-eval/index.js +356 -345
  40. package/dist/builder-eval/index.js.map +1 -1
  41. package/dist/calibration-CNWWA6K8.js +94 -0
  42. package/dist/calibration-CNWWA6K8.js.map +1 -0
  43. package/dist/campaign/index.d.ts +5 -6381
  44. package/dist/campaign/index.js +3 -213
  45. package/dist/campaign-CBKZvQ1H.js +3885 -0
  46. package/dist/campaign-CBKZvQ1H.js.map +1 -0
  47. package/dist/cli.d.ts +1 -1
  48. package/dist/cli.js +164 -175
  49. package/dist/cli.js.map +1 -1
  50. package/dist/client-C97NMzqi.d.ts +581 -0
  51. package/dist/client-C97NMzqi.d.ts.map +1 -0
  52. package/dist/client-CYzbdJOZ.js +637 -0
  53. package/dist/client-CYzbdJOZ.js.map +1 -0
  54. package/dist/code-agent-session-BjkMTQ7H.js +1390 -0
  55. package/dist/code-agent-session-BjkMTQ7H.js.map +1 -0
  56. package/dist/code-agent-session-DqqgOJaz.d.ts +143 -0
  57. package/dist/code-agent-session-DqqgOJaz.d.ts.map +1 -0
  58. package/dist/concurrency-DIxRZF_J.js +85 -0
  59. package/dist/concurrency-DIxRZF_J.js.map +1 -0
  60. package/dist/contract/index.d.ts +645 -5565
  61. package/dist/contract/index.d.ts.map +1 -0
  62. package/dist/contract/index.js +1730 -1938
  63. package/dist/contract/index.js.map +1 -1
  64. package/dist/control.d.ts +3 -1030
  65. package/dist/control.js +2 -33
  66. package/dist/cost-ledger-DIgQUFZZ.js +801 -0
  67. package/dist/cost-ledger-DIgQUFZZ.js.map +1 -0
  68. package/dist/cost-ledger-Dye6jCgg.d.ts +236 -0
  69. package/dist/cost-ledger-Dye6jCgg.d.ts.map +1 -0
  70. package/dist/dataset-BvtnC8Dc.d.ts +115 -0
  71. package/dist/dataset-BvtnC8Dc.d.ts.map +1 -0
  72. package/dist/default-registry-C-vFCSEc.js +2579 -0
  73. package/dist/default-registry-C-vFCSEc.js.map +1 -0
  74. package/dist/default-registry-CNPo-Vsb.d.ts +540 -0
  75. package/dist/default-registry-CNPo-Vsb.d.ts.map +1 -0
  76. package/dist/emitter-CPBAhxum.js +266 -0
  77. package/dist/emitter-CPBAhxum.js.map +1 -0
  78. package/dist/emitter-DGQGoLyj.d.ts +113 -0
  79. package/dist/emitter-DGQGoLyj.d.ts.map +1 -0
  80. package/dist/errors-8YnH8WlF.js +64 -0
  81. package/dist/errors-8YnH8WlF.js.map +1 -0
  82. package/dist/errors-CEk209JS.d.ts +76 -0
  83. package/dist/errors-CEk209JS.d.ts.map +1 -0
  84. package/dist/eval-campaign-DEm6c8ru.js +349 -0
  85. package/dist/eval-campaign-DEm6c8ru.js.map +1 -0
  86. package/dist/execution-tracks-CpgFPpS5.js +93 -0
  87. package/dist/execution-tracks-CpgFPpS5.js.map +1 -0
  88. package/dist/exporters-q9iL-2Jf.js +148 -0
  89. package/dist/exporters-q9iL-2Jf.js.map +1 -0
  90. package/dist/extract-usage-BrQ8mCLX.js +155 -0
  91. package/dist/extract-usage-BrQ8mCLX.js.map +1 -0
  92. package/dist/failure-cluster-CqcvCcdR.d.ts +59 -0
  93. package/dist/failure-cluster-CqcvCcdR.d.ts.map +1 -0
  94. package/dist/feedback-trajectory-CVaeREXV.d.ts +340 -0
  95. package/dist/feedback-trajectory-CVaeREXV.d.ts.map +1 -0
  96. package/dist/fuzz.d.ts +320 -646
  97. package/dist/fuzz.d.ts.map +1 -0
  98. package/dist/fuzz.js +670 -618
  99. package/dist/fuzz.js.map +1 -1
  100. package/dist/hf-dataset-DBJXXoY1.js +763 -0
  101. package/dist/hf-dataset-DBJXXoY1.js.map +1 -0
  102. package/dist/hosted/index.d.ts +11 -831
  103. package/dist/hosted/index.d.ts.map +1 -0
  104. package/dist/hosted/index.js +2 -37
  105. package/dist/index-2JJSA6-r2.d.ts +926 -0
  106. package/dist/index-2JJSA6-r2.d.ts.map +1 -0
  107. package/dist/index-6N0aYmpW.d.ts +217 -0
  108. package/dist/index-6N0aYmpW.d.ts.map +1 -0
  109. package/dist/index-BAvgST_9.d.ts +131 -0
  110. package/dist/index-BAvgST_9.d.ts.map +1 -0
  111. package/dist/index-BvnJuTGD.d.ts +68 -0
  112. package/dist/index-BvnJuTGD.d.ts.map +1 -0
  113. package/dist/index-C61Wi7yg.d.ts +547 -0
  114. package/dist/index-C61Wi7yg.d.ts.map +1 -0
  115. package/dist/index-CAPUUKaM.d.ts +335 -0
  116. package/dist/index-CAPUUKaM.d.ts.map +1 -0
  117. package/dist/index-DE5fb3EC.d.ts +2244 -0
  118. package/dist/index-DE5fb3EC.d.ts.map +1 -0
  119. package/dist/index-DSC51roc.d.ts +102 -0
  120. package/dist/index-DSC51roc.d.ts.map +1 -0
  121. package/dist/index.d.ts +3755 -15555
  122. package/dist/index.d.ts.map +1 -0
  123. package/dist/index.js +11182 -11216
  124. package/dist/index.js.map +1 -1
  125. package/dist/integrity-BzRbCHzi.js +424 -0
  126. package/dist/integrity-BzRbCHzi.js.map +1 -0
  127. package/dist/integrity-rmVhXWA7.d.ts +61 -0
  128. package/dist/integrity-rmVhXWA7.d.ts.map +1 -0
  129. package/dist/judge-calibration-DFtEMlde.d.ts +146 -0
  130. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  131. package/dist/ledger-core/index.d.ts +2 -0
  132. package/dist/ledger-core/index.js +2 -0
  133. package/dist/ledger-core-DtZz1RG0.js +388 -0
  134. package/dist/ledger-core-DtZz1RG0.js.map +1 -0
  135. package/dist/llm-client--GR4JbZE.js +687 -0
  136. package/dist/llm-client--GR4JbZE.js.map +1 -0
  137. package/dist/llm-client-B_nIBlYo.d.ts +290 -0
  138. package/dist/llm-client-B_nIBlYo.d.ts.map +1 -0
  139. package/dist/matrix/index.d.ts +3 -155
  140. package/dist/matrix/index.js +2 -8
  141. package/dist/matrix-BzQnu2S6.js +270 -0
  142. package/dist/matrix-BzQnu2S6.js.map +1 -0
  143. package/dist/meta-eval/index.d.ts +4 -1027
  144. package/dist/meta-eval/index.js +393 -390
  145. package/dist/meta-eval/index.js.map +1 -1
  146. package/dist/metrics-C9YY1OcL.js +239 -0
  147. package/dist/metrics-C9YY1OcL.js.map +1 -0
  148. package/dist/mint-yN2M2eh0.js +201 -0
  149. package/dist/mint-yN2M2eh0.js.map +1 -0
  150. package/dist/multi-layer-verifier-BHY1gWAc.d.ts +138 -0
  151. package/dist/multi-layer-verifier-BHY1gWAc.d.ts.map +1 -0
  152. package/dist/multishot/index.d.ts +273 -480
  153. package/dist/multishot/index.d.ts.map +1 -0
  154. package/dist/multishot/index.js +595 -548
  155. package/dist/multishot/index.js.map +1 -1
  156. package/dist/off-policy-DvgzvtIx.js +220 -0
  157. package/dist/off-policy-DvgzvtIx.js.map +1 -0
  158. package/dist/off-policy-mskQw8Mb.d.ts +153 -0
  159. package/dist/off-policy-mskQw8Mb.d.ts.map +1 -0
  160. package/dist/openapi.json +1 -1
  161. package/dist/opencode-sqlite-BGrHeDu3.js +318 -0
  162. package/dist/opencode-sqlite-BGrHeDu3.js.map +1 -0
  163. package/dist/outcome-store-BYHIuO0e.d.ts +65 -0
  164. package/dist/outcome-store-BYHIuO0e.d.ts.map +1 -0
  165. package/dist/outcome-store-ChBKlTd_.js +75 -0
  166. package/dist/outcome-store-ChBKlTd_.js.map +1 -0
  167. package/dist/paired-arms-D9D0wXj2.js +260 -0
  168. package/dist/paired-arms-D9D0wXj2.js.map +1 -0
  169. package/dist/pipelines/index.d.ts +95 -532
  170. package/dist/pipelines/index.d.ts.map +1 -0
  171. package/dist/pipelines/index.js +497 -478
  172. package/dist/pipelines/index.js.map +1 -1
  173. package/dist/pre-registration-DakwTRXk.js +96 -0
  174. package/dist/pre-registration-DakwTRXk.js.map +1 -0
  175. package/dist/propose-review-control-LhIWGzHl.js +1458 -0
  176. package/dist/propose-review-control-LhIWGzHl.js.map +1 -0
  177. package/dist/query-CJ_DX8vl.d.ts +28 -0
  178. package/dist/query-CJ_DX8vl.d.ts.map +1 -0
  179. package/dist/query-Di7eEQ79.js +83 -0
  180. package/dist/query-Di7eEQ79.js.map +1 -0
  181. package/dist/raw-provider-sink-BQd7mzyT.js +201 -0
  182. package/dist/raw-provider-sink-BQd7mzyT.js.map +1 -0
  183. package/dist/raw-provider-sink-BU29Sh8h.d.ts +134 -0
  184. package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +1 -0
  185. package/dist/redact-7Aq1ukl-.js +107 -0
  186. package/dist/redact-7Aq1ukl-.js.map +1 -0
  187. package/dist/release-report-Cz9NKH39.js +603 -0
  188. package/dist/release-report-Cz9NKH39.js.map +1 -0
  189. package/dist/release-report-mvB2G4_J.d.ts +244 -0
  190. package/dist/release-report-mvB2G4_J.d.ts.map +1 -0
  191. package/dist/replay-GnyotH0J.js +1741 -0
  192. package/dist/replay-GnyotH0J.js.map +1 -0
  193. package/dist/replay-OoidtG1E.d.ts +749 -0
  194. package/dist/replay-OoidtG1E.d.ts.map +1 -0
  195. package/dist/reporting.d.ts +6 -1312
  196. package/dist/reporting.js +6 -51
  197. package/dist/researcher-CwTdwXG1.d.ts +314 -0
  198. package/dist/researcher-CwTdwXG1.d.ts.map +1 -0
  199. package/dist/reward-hacking-eAnOsynk.d.ts +224 -0
  200. package/dist/reward-hacking-eAnOsynk.d.ts.map +1 -0
  201. package/dist/reward-hacking-qipEpKvY.js +596 -0
  202. package/dist/reward-hacking-qipEpKvY.js.map +1 -0
  203. package/dist/reward-nw2xZGZG.js +137 -0
  204. package/dist/reward-nw2xZGZG.js.map +1 -0
  205. package/dist/rl.d.ts +760 -4010
  206. package/dist/rl.d.ts.map +1 -0
  207. package/dist/rl.js +2325 -1958
  208. package/dist/rl.js.map +1 -1
  209. package/dist/rolldown-runtime-8H4AJuhK.js +14 -0
  210. package/dist/rollout/index.d.ts +3 -2087
  211. package/dist/rollout/index.js +8 -168
  212. package/dist/rollout-BOYjemfR.js +624 -0
  213. package/dist/rollout-BOYjemfR.js.map +1 -0
  214. package/dist/rubric-predictive-validity-B3xmbmS1.js +141 -0
  215. package/dist/rubric-predictive-validity-B3xmbmS1.js.map +1 -0
  216. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts +74 -0
  217. package/dist/rubric-predictive-validity-Ku_clp_1.d.ts.map +1 -0
  218. package/dist/run-evidence-ClRX_8A9.d.ts +225 -0
  219. package/dist/run-evidence-ClRX_8A9.d.ts.map +1 -0
  220. package/dist/run-record-BuoE80Dq.js +467 -0
  221. package/dist/{chunk-56TAVBOK.js.map → run-record-BuoE80Dq.js.map} +1 -1
  222. package/dist/run-record-CnZu_gjl.d.ts +357 -0
  223. package/dist/run-record-CnZu_gjl.d.ts.map +1 -0
  224. package/dist/run-score-iEEAWiBY.js +41 -0
  225. package/dist/run-score-iEEAWiBY.js.map +1 -0
  226. package/dist/runtime-trajectory-1gyaTOoC.js +93 -0
  227. package/dist/runtime-trajectory-1gyaTOoC.js.map +1 -0
  228. package/dist/runtime-trajectory-BvSZcCHD.d.ts +50 -0
  229. package/dist/runtime-trajectory-BvSZcCHD.d.ts.map +1 -0
  230. package/dist/schema-BtVldJ3T.d.ts +206 -0
  231. package/dist/schema-BtVldJ3T.d.ts.map +1 -0
  232. package/dist/schema-C6DW4ZHR.js +821 -0
  233. package/dist/schema-C6DW4ZHR.js.map +1 -0
  234. package/dist/schema-CRhEY1SO.js +69 -0
  235. package/dist/schema-CRhEY1SO.js.map +1 -0
  236. package/dist/schema-Cef2cFmb.d.ts +408 -0
  237. package/dist/schema-Cef2cFmb.d.ts.map +1 -0
  238. package/dist/semantic-concept-judge-B6cWNJ2K.js +725 -0
  239. package/dist/semantic-concept-judge-B6cWNJ2K.js.map +1 -0
  240. package/dist/sequential-Br0mAPHA.js +148 -0
  241. package/dist/sequential-Br0mAPHA.js.map +1 -0
  242. package/dist/sequential-CYwq6Ff_.d.ts +141 -0
  243. package/dist/sequential-CYwq6Ff_.d.ts.map +1 -0
  244. package/dist/series-convergence-CjO2QdRW.js +43 -0
  245. package/dist/series-convergence-CjO2QdRW.js.map +1 -0
  246. package/dist/series-convergence-ofsqPWhs.d.ts +35 -0
  247. package/dist/series-convergence-ofsqPWhs.d.ts.map +1 -0
  248. package/dist/server-m5D9cvnG.js +1040 -0
  249. package/dist/server-m5D9cvnG.js.map +1 -0
  250. package/dist/skill-usage-C_pXm7lP.d.ts +534 -0
  251. package/dist/skill-usage-C_pXm7lP.d.ts.map +1 -0
  252. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts +1739 -0
  253. package/dist/skillopt-optimization-method-B7wX7XkF.d.ts.map +1 -0
  254. package/dist/skillopt-optimization-method-D4ODwFVV.js +7712 -0
  255. package/dist/skillopt-optimization-method-D4ODwFVV.js.map +1 -0
  256. package/dist/statistics-Cmj6nynr.d.ts +514 -0
  257. package/dist/statistics-Cmj6nynr.d.ts.map +1 -0
  258. package/dist/statistics-CnnxdpOg.js +1437 -0
  259. package/dist/statistics-CnnxdpOg.js.map +1 -0
  260. package/dist/store-CT9YIIve.d.ts +117 -0
  261. package/dist/store-CT9YIIve.d.ts.map +1 -0
  262. package/dist/store-CxJry_cs.d.ts +229 -0
  263. package/dist/store-CxJry_cs.d.ts.map +1 -0
  264. package/dist/storyboard/index.d.ts +84 -203
  265. package/dist/storyboard/index.d.ts.map +1 -0
  266. package/dist/storyboard/index.js +609 -542
  267. package/dist/storyboard/index.js.map +1 -1
  268. package/dist/summary-report-BNs5nmXI.js +862 -0
  269. package/dist/summary-report-BNs5nmXI.js.map +1 -0
  270. package/dist/summary-report-Cj9gdw4i.d.ts +407 -0
  271. package/dist/summary-report-Cj9gdw4i.d.ts.map +1 -0
  272. package/dist/supervisor-run/index.d.ts +2 -959
  273. package/dist/supervisor-run/index.js +2 -65
  274. package/dist/supervisor-run-_lnTLM3z.js +1679 -0
  275. package/dist/supervisor-run-_lnTLM3z.js.map +1 -0
  276. package/dist/task-failure-attributes-CQZlB3et.js +311 -0
  277. package/dist/task-failure-attributes-CQZlB3et.js.map +1 -0
  278. package/dist/test-graded-scenario-BsqWLmPt.js +318 -0
  279. package/dist/test-graded-scenario-BsqWLmPt.js.map +1 -0
  280. package/dist/test-graded-scenario-D1TaI2va.d.ts +141 -0
  281. package/dist/test-graded-scenario-D1TaI2va.d.ts.map +1 -0
  282. package/dist/tools-BmuN627J.js +1085 -0
  283. package/dist/tools-BmuN627J.js.map +1 -0
  284. package/dist/trace-attributes.d.ts +2 -52
  285. package/dist/trace-attributes.js +131 -61
  286. package/dist/trace-attributes.js.map +1 -1
  287. package/dist/traces.d.ts +12 -2365
  288. package/dist/traces.js +12 -252
  289. package/dist/trajectory-D_7rLrvE.js +56 -0
  290. package/dist/trajectory-D_7rLrvE.js.map +1 -0
  291. package/dist/types-DGsxbAEd.d.ts +387 -0
  292. package/dist/types-DGsxbAEd.d.ts.map +1 -0
  293. package/dist/types-k9tZGKUg.d.ts +640 -0
  294. package/dist/types-k9tZGKUg.d.ts.map +1 -0
  295. package/dist/verdict-Dps8_okt.d.ts +37 -0
  296. package/dist/verdict-Dps8_okt.d.ts.map +1 -0
  297. package/dist/wire/index.d.ts +702 -1173
  298. package/dist/wire/index.d.ts.map +1 -0
  299. package/dist/wire/index.js +2 -81
  300. package/package.json +17 -9
  301. package/dist/benchmarks/index.js.map +0 -1
  302. package/dist/campaign/index.js.map +0 -1
  303. package/dist/chunk-2QU3YOPR.js +0 -7374
  304. package/dist/chunk-2QU3YOPR.js.map +0 -1
  305. package/dist/chunk-3OCR4R5I.js +0 -728
  306. package/dist/chunk-3OCR4R5I.js.map +0 -1
  307. package/dist/chunk-3RF76KTD.js +0 -84
  308. package/dist/chunk-3RF76KTD.js.map +0 -1
  309. package/dist/chunk-56TAVBOK.js +0 -698
  310. package/dist/chunk-5DTSBUL2.js +0 -159
  311. package/dist/chunk-5DTSBUL2.js.map +0 -1
  312. package/dist/chunk-7FO3TNPI.js +0 -232
  313. package/dist/chunk-7FO3TNPI.js.map +0 -1
  314. package/dist/chunk-7ZZMD7UK.js +0 -386
  315. package/dist/chunk-7ZZMD7UK.js.map +0 -1
  316. package/dist/chunk-BOD4O7OF.js +0 -40
  317. package/dist/chunk-BOD4O7OF.js.map +0 -1
  318. package/dist/chunk-BSO5JDQH.js +0 -2335
  319. package/dist/chunk-BSO5JDQH.js.map +0 -1
  320. package/dist/chunk-C6LXANRU.js +0 -1550
  321. package/dist/chunk-C6LXANRU.js.map +0 -1
  322. package/dist/chunk-DODXQREJ.js +0 -752
  323. package/dist/chunk-DODXQREJ.js.map +0 -1
  324. package/dist/chunk-DRYIUNWY.js +0 -622
  325. package/dist/chunk-DRYIUNWY.js.map +0 -1
  326. package/dist/chunk-E7QXT7SX.js +0 -183
  327. package/dist/chunk-E7QXT7SX.js.map +0 -1
  328. package/dist/chunk-EG66UGL4.js +0 -341
  329. package/dist/chunk-EG66UGL4.js.map +0 -1
  330. package/dist/chunk-FXTVJPYD.js +0 -576
  331. package/dist/chunk-FXTVJPYD.js.map +0 -1
  332. package/dist/chunk-G7MGMCZD.js +0 -153
  333. package/dist/chunk-G7MGMCZD.js.map +0 -1
  334. package/dist/chunk-GGE4NNQT.js +0 -65
  335. package/dist/chunk-GGE4NNQT.js.map +0 -1
  336. package/dist/chunk-H23X7XKK.js +0 -181
  337. package/dist/chunk-H23X7XKK.js.map +0 -1
  338. package/dist/chunk-HHWE3POT.js +0 -94
  339. package/dist/chunk-HHWE3POT.js.map +0 -1
  340. package/dist/chunk-HPWUNB47.js +0 -289
  341. package/dist/chunk-HPWUNB47.js.map +0 -1
  342. package/dist/chunk-IYCLP2N2.js +0 -766
  343. package/dist/chunk-IYCLP2N2.js.map +0 -1
  344. package/dist/chunk-JHCHEVET.js +0 -274
  345. package/dist/chunk-JHCHEVET.js.map +0 -1
  346. package/dist/chunk-JQSF5DQT.js +0 -701
  347. package/dist/chunk-JQSF5DQT.js.map +0 -1
  348. package/dist/chunk-K4DBDHLK.js +0 -158
  349. package/dist/chunk-K4DBDHLK.js.map +0 -1
  350. package/dist/chunk-K6N6XJJX.js +0 -306
  351. package/dist/chunk-K6N6XJJX.js.map +0 -1
  352. package/dist/chunk-M4YBQKIJ.js +0 -1040
  353. package/dist/chunk-M4YBQKIJ.js.map +0 -1
  354. package/dist/chunk-MA6HLL3S.js +0 -65
  355. package/dist/chunk-MA6HLL3S.js.map +0 -1
  356. package/dist/chunk-MAZ26DC7.js +0 -99
  357. package/dist/chunk-MAZ26DC7.js.map +0 -1
  358. package/dist/chunk-NPCTHQIO.js +0 -91
  359. package/dist/chunk-NPCTHQIO.js.map +0 -1
  360. package/dist/chunk-NY44NC4A.js +0 -1056
  361. package/dist/chunk-NY44NC4A.js.map +0 -1
  362. package/dist/chunk-OIUOT4QD.js +0 -44
  363. package/dist/chunk-OIUOT4QD.js.map +0 -1
  364. package/dist/chunk-ONWEPEDO.js +0 -57
  365. package/dist/chunk-ONWEPEDO.js.map +0 -1
  366. package/dist/chunk-OWN5NPMC.js +0 -152
  367. package/dist/chunk-OWN5NPMC.js.map +0 -1
  368. package/dist/chunk-P6FYH6K4.js +0 -1161
  369. package/dist/chunk-P6FYH6K4.js.map +0 -1
  370. package/dist/chunk-PC4UYEBM.js +0 -166
  371. package/dist/chunk-PC4UYEBM.js.map +0 -1
  372. package/dist/chunk-PC5DOSM7.js +0 -579
  373. package/dist/chunk-PC5DOSM7.js.map +0 -1
  374. package/dist/chunk-PXE2VKMX.js +0 -140
  375. package/dist/chunk-PXE2VKMX.js.map +0 -1
  376. package/dist/chunk-PZ5AY32C.js +0 -10
  377. package/dist/chunk-PZ5AY32C.js.map +0 -1
  378. package/dist/chunk-QB6BDBP2.js +0 -4464
  379. package/dist/chunk-QB6BDBP2.js.map +0 -1
  380. package/dist/chunk-RXHCETDZ.js +0 -536
  381. package/dist/chunk-RXHCETDZ.js.map +0 -1
  382. package/dist/chunk-RZTMDUO7.js +0 -49
  383. package/dist/chunk-RZTMDUO7.js.map +0 -1
  384. package/dist/chunk-SFLLL76A.js +0 -669
  385. package/dist/chunk-SFLLL76A.js.map +0 -1
  386. package/dist/chunk-SZLVEKMJ.js +0 -1446
  387. package/dist/chunk-SZLVEKMJ.js.map +0 -1
  388. package/dist/chunk-T4SQEITX.js +0 -95
  389. package/dist/chunk-T4SQEITX.js.map +0 -1
  390. package/dist/chunk-T6RLYGAD.js +0 -158
  391. package/dist/chunk-T6RLYGAD.js.map +0 -1
  392. package/dist/chunk-TJVT4QFF.js +0 -911
  393. package/dist/chunk-TJVT4QFF.js.map +0 -1
  394. package/dist/chunk-TQ7LNKZ3.js +0 -136
  395. package/dist/chunk-TQ7LNKZ3.js.map +0 -1
  396. package/dist/chunk-U4L7JRPZ.js +0 -1706
  397. package/dist/chunk-U4L7JRPZ.js.map +0 -1
  398. package/dist/chunk-U4PHLT2N.js +0 -419
  399. package/dist/chunk-U4PHLT2N.js.map +0 -1
  400. package/dist/chunk-VCZ5FQYW.js +0 -928
  401. package/dist/chunk-VCZ5FQYW.js.map +0 -1
  402. package/dist/chunk-VI2UW6B6.js +0 -162
  403. package/dist/chunk-VI2UW6B6.js.map +0 -1
  404. package/dist/chunk-VQMK5FMP.js +0 -247
  405. package/dist/chunk-VQMK5FMP.js.map +0 -1
  406. package/dist/chunk-WGXIEX7P.js +0 -116
  407. package/dist/chunk-WGXIEX7P.js.map +0 -1
  408. package/dist/chunk-WVATSFCP.js +0 -1553
  409. package/dist/chunk-WVATSFCP.js.map +0 -1
  410. package/dist/chunk-X4YIBDER.js +0 -1662
  411. package/dist/chunk-X4YIBDER.js.map +0 -1
  412. package/dist/chunk-YQN4ICPP.js +0 -355
  413. package/dist/chunk-YQN4ICPP.js.map +0 -1
  414. package/dist/chunk-ZET2UAYW.js +0 -89
  415. package/dist/chunk-ZET2UAYW.js.map +0 -1
  416. package/dist/chunk-ZHTZ4EYI.js +0 -1212
  417. package/dist/chunk-ZHTZ4EYI.js.map +0 -1
  418. package/dist/control.js.map +0 -1
  419. package/dist/hosted/index.js.map +0 -1
  420. package/dist/matrix/index.js.map +0 -1
  421. package/dist/reporting.js.map +0 -1
  422. package/dist/rollout/index.js.map +0 -1
  423. package/dist/run-campaign-OJJ7CZF4.js +0 -18
  424. package/dist/run-campaign-OJJ7CZF4.js.map +0 -1
  425. package/dist/supervisor-run/index.js.map +0 -1
  426. package/dist/traces.js.map +0 -1
  427. package/dist/wire/index.js.map +0 -1
@@ -0,0 +1,1040 @@
1
+ import { i as CostLedger } from "./cost-ledger-DIgQUFZZ.js";
2
+ import { a as assertLlmRoute, c as callLlmJson, f as maximumChargeForLlmRequest, i as LlmRouteAssertionError, l as costReceiptFromLlm, u as costReceiptFromLlmError } from "./llm-client--GR4JbZE.js";
3
+ import { z } from "zod";
4
+ import { readFileSync } from "node:fs";
5
+ import { dirname, resolve } from "node:path";
6
+ import { OpenAPIRegistry, OpenApiGeneratorV31, extendZodWithOpenApi } from "@asteasolutions/zod-to-openapi";
7
+ import { fileURLToPath } from "node:url";
8
+ import { serve } from "@hono/node-server";
9
+ import { Hono } from "hono";
10
+ import { cors } from "hono/cors";
11
+ //#region src/wire/schemas.ts
12
+ /**
13
+ * Wire-protocol schemas.
14
+ *
15
+ * These Zod schemas are the contract between the agent-eval runtime and
16
+ * any non-TypeScript client (Python, Rust, Go, …). They get rendered to
17
+ * OpenAPI by `wire/openapi.ts` and code-generators consume that spec to
18
+ * produce typed clients in other languages.
19
+ *
20
+ * Rule: if it's not in this file, it isn't on the wire. Keep names and
21
+ * shapes self-explanatory — every field has a `.describe()` so the
22
+ * generated docs are useful without reading the source.
23
+ */
24
+ extendZodWithOpenApi(z);
25
+ const RubricDimensionSchema = z.object({
26
+ id: z.string().min(1).describe("Short stable id like \"buyer_quality\" — used as the key in scoring output."),
27
+ description: z.string().min(1).describe("One-line plain-English meaning. Read by humans reviewing low scores."),
28
+ weight: z.number().min(0).default(1).describe("Relative weight in the composite score. Default 1; 0 disables."),
29
+ min: z.number().default(0).describe("Lower bound of valid score for this dimension."),
30
+ max: z.number().default(1).describe("Upper bound of valid score for this dimension.")
31
+ }).openapi("RubricDimension");
32
+ const FailureModeSchema = z.object({
33
+ id: z.string().min(1).describe("Short stable id like \"ai-cadence\" — used in detection lists."),
34
+ description: z.string().min(1).describe("Plain-English description of the failure pattern.")
35
+ }).openapi("FailureMode");
36
+ const RubricSchema = z.object({
37
+ name: z.string().min(1).describe("Stable name like \"anti-slop\" — used by clients to invoke this rubric."),
38
+ description: z.string().min(1).describe("What this rubric measures. Shown in /v1/rubrics listing."),
39
+ systemPrompt: z.string().min(1).describe("Instructs the judging LLM. Should explain the persona (e.g. \"senior engineer reviewing voice\"), what to score on, and what to return."),
40
+ dimensions: z.array(RubricDimensionSchema).min(1).describe("Scoring axes. The composite score is a weighted sum of these."),
41
+ failureModes: z.array(FailureModeSchema).default([]).describe("Patterns to detect; each detected mode appears in the result.failureModes list."),
42
+ wins: z.array(FailureModeSchema).default([]).describe("Positive patterns; each detected one appears in the result.wins list.")
43
+ }).openapi("Rubric");
44
+ const JudgeRequestSchema = z.object({
45
+ rubricName: z.string().optional().describe("Use a built-in rubric by name. Mutually exclusive with `rubric`."),
46
+ rubric: RubricSchema.optional().describe("Inline rubric definition. Mutually exclusive with `rubricName`."),
47
+ content: z.string().min(1).describe("The text being judged — a tweet, a blog post, a code snippet, anything stringly."),
48
+ context: z.record(z.string(), z.unknown()).optional().describe("Free-form metadata for the rubric to use — analytics, source URL, author, etc. Surfaced to the LLM."),
49
+ model: z.string().optional().describe("Override the judge model configured by the server.")
50
+ }).refine((v) => Boolean(v.rubricName) !== Boolean(v.rubric), { message: "Provide exactly one of `rubricName` or `rubric`." }).openapi("JudgeRequest");
51
+ const JudgeResultSchema = z.object({
52
+ composite: z.number().min(0).max(1).describe("Weighted combination of dimension scores in 0..1. The single number to gate on."),
53
+ dimensions: z.record(z.string(), z.number()).describe("Per-dimension score, keyed by RubricDimension.id."),
54
+ failureModes: z.array(z.string()).default([]).describe("Failure-mode ids detected in the content (subset of rubric.failureModes ids)."),
55
+ wins: z.array(z.string()).default([]).describe("Win ids detected in the content (subset of rubric.wins ids)."),
56
+ rationale: z.string().describe("Plain-English explanation of the score. Surfaced to the human reviewer."),
57
+ rubricVersion: z.string().describe("Stable hash of the rubric used. Scores are only comparable across runs when this matches."),
58
+ model: z.string().describe("Model that produced the judgement, for reproducibility."),
59
+ durationMs: z.number().int().nonnegative().describe("End-to-end wall time for this call.")
60
+ }).openapi("JudgeResult");
61
+ const RubricInfoSchema = z.object({
62
+ name: z.string().describe("Pass this to /v1/judge as `rubricName`."),
63
+ description: z.string().describe("What this rubric measures."),
64
+ dimensions: z.array(z.object({
65
+ id: z.string(),
66
+ description: z.string(),
67
+ weight: z.number()
68
+ })).describe("The scoring axes this rubric uses, with weights."),
69
+ failureModes: z.array(z.string()).default([]).describe("Failure-mode ids this rubric detects."),
70
+ rubricVersion: z.string().describe("Stable hash — match this to compare scores across runs.")
71
+ }).openapi("RubricInfo");
72
+ const ListRubricsResponseSchema = z.object({ rubrics: z.array(RubricInfoSchema) }).openapi("ListRubricsResponse");
73
+ const VersionResponseSchema = z.object({
74
+ package: z.string().describe("Package name (always \"@tangle-network/agent-eval\")."),
75
+ version: z.string().describe("Semver of the running server. Match your client to this."),
76
+ wireVersion: z.string().describe("Wire-protocol semver. Bumps separately from package version when the schema changes."),
77
+ apiSurface: z.array(z.string()).describe("List of supported method names.")
78
+ }).openapi("VersionResponse");
79
+ const HealthResponseSchema = z.object({
80
+ status: z.literal("ok"),
81
+ uptimeSec: z.number()
82
+ }).openapi("HealthResponse");
83
+ /**
84
+ * Minimal `TraceEvent` shape that the production runtime emits.
85
+ * Matches `trace/schema.ts` `TraceEvent` but is duplicated here as a
86
+ * wire schema so non-TypeScript clients can validate without depending
87
+ * on internal types.
88
+ */
89
+ const TraceEventSchema = z.object({
90
+ eventId: z.string().min(1).describe("Stable id for the event. Use ULID or UUID."),
91
+ runId: z.string().min(1).describe("Run this event belongs to."),
92
+ spanId: z.string().optional().describe("Span that emitted the event, if any."),
93
+ kind: z.enum([
94
+ "log",
95
+ "error",
96
+ "budget_decrement",
97
+ "budget_breach",
98
+ "state_mutation",
99
+ "policy_violation",
100
+ "redaction_applied",
101
+ "custom"
102
+ ]).describe("Coarse event category — matches the TraceSchema v1 EventKind enum."),
103
+ timestamp: z.number().int().nonnegative().describe("Unix millis. Must be monotonically non-decreasing within a span."),
104
+ payload: z.record(z.string(), z.unknown()).describe("Free-form payload — the runtime owns the shape.")
105
+ }).openapi("TraceEvent");
106
+ const TracesIngestRequestSchema = z.object({ events: z.array(TraceEventSchema).min(1).max(1e4).describe("Batch of events. Max 10k per call — bigger streams should be chunked.") }).openapi("TracesIngestRequest");
107
+ const TracesIngestResponseSchema = z.object({
108
+ accepted: z.number().int().nonnegative().describe("Number of events persisted."),
109
+ rejected: z.number().int().nonnegative().describe("Number of events the store refused — see `errors[]` for reasons."),
110
+ errors: z.array(z.object({
111
+ eventId: z.string().describe("Event id this error applies to."),
112
+ message: z.string().describe("Why the event was rejected.")
113
+ })).default([])
114
+ }).openapi("TracesIngestResponse");
115
+ const FeedbackLabelSchema = z.object({
116
+ id: z.string().optional(),
117
+ source: z.enum([
118
+ "user",
119
+ "judge",
120
+ "environment",
121
+ "metric",
122
+ "policy",
123
+ "system"
124
+ ]),
125
+ kind: z.enum([
126
+ "approve",
127
+ "reject",
128
+ "select",
129
+ "edit",
130
+ "rank",
131
+ "rate",
132
+ "comment",
133
+ "metric_outcome",
134
+ "policy_block",
135
+ "revision_request"
136
+ ]),
137
+ value: z.unknown(),
138
+ reason: z.string().optional(),
139
+ severity: z.enum([
140
+ "info",
141
+ "warning",
142
+ "error",
143
+ "critical"
144
+ ]).optional(),
145
+ createdAt: z.string().describe("ISO-8601 UTC."),
146
+ metadata: z.record(z.string(), z.unknown()).optional()
147
+ }).openapi("FeedbackLabel");
148
+ const FeedbackAttemptSchema = z.object({
149
+ id: z.string().min(1),
150
+ stepIndex: z.number().int().nonnegative(),
151
+ artifactType: z.enum([
152
+ "text",
153
+ "code",
154
+ "plan",
155
+ "research",
156
+ "action",
157
+ "ui",
158
+ "decision",
159
+ "data",
160
+ "other"
161
+ ]),
162
+ artifact: z.unknown(),
163
+ options: z.array(z.unknown()).optional(),
164
+ proposedAction: z.object({
165
+ type: z.string(),
166
+ risk: z.enum([
167
+ "low",
168
+ "medium",
169
+ "high"
170
+ ]).optional(),
171
+ costUsd: z.number().optional(),
172
+ externalSideEffect: z.boolean().optional(),
173
+ requiresApproval: z.boolean().optional(),
174
+ metadata: z.record(z.string(), z.unknown()).optional()
175
+ }).optional(),
176
+ feedback: z.array(FeedbackLabelSchema).optional(),
177
+ createdAt: z.string(),
178
+ metadata: z.record(z.string(), z.unknown()).optional()
179
+ }).openapi("FeedbackAttempt");
180
+ const FeedbackTrajectorySchema = z.object({
181
+ id: z.string().min(1).describe("Stable id; idempotency key for the trajectory."),
182
+ projectId: z.string().optional(),
183
+ scenarioId: z.string().optional(),
184
+ task: z.object({
185
+ intent: z.string().min(1),
186
+ context: z.unknown().optional()
187
+ }),
188
+ attempts: z.array(FeedbackAttemptSchema).default([]),
189
+ labels: z.array(FeedbackLabelSchema).default([]),
190
+ outcome: z.object({
191
+ success: z.boolean().optional(),
192
+ score: z.number().optional(),
193
+ metrics: z.record(z.string(), z.number()).optional(),
194
+ costUsd: z.number().optional(),
195
+ detail: z.string().optional(),
196
+ observedAt: z.string().optional(),
197
+ metadata: z.record(z.string(), z.unknown()).optional()
198
+ }).optional(),
199
+ split: z.enum([
200
+ "train",
201
+ "dev",
202
+ "test",
203
+ "holdout"
204
+ ]).optional(),
205
+ tags: z.record(z.string(), z.string()).optional(),
206
+ createdAt: z.string().describe("ISO-8601 UTC."),
207
+ updatedAt: z.string().optional(),
208
+ metadata: z.record(z.string(), z.unknown()).optional()
209
+ }).openapi("FeedbackTrajectory");
210
+ const FeedbackIngestResponseSchema = z.object({
211
+ id: z.string().describe("Trajectory id that was persisted."),
212
+ persisted: z.boolean().describe("True when the trajectory was saved (idempotent on id).")
213
+ }).openapi("FeedbackIngestResponse");
214
+ const ErrorResponseSchema = z.object({ error: z.object({
215
+ code: z.string().describe("Machine-readable code: \"validation_error\", \"rubric_not_found\", \"judge_error\"."),
216
+ message: z.string().describe("Human-readable message."),
217
+ details: z.unknown().optional().describe("Optional structured detail.")
218
+ }).describe("Errors are always wrapped in this shape across all endpoints.") }).openapi("ErrorResponse");
219
+ /**
220
+ * Bump on any breaking change to a request/response schema.
221
+ * Non-breaking (additive) changes don't require a bump.
222
+ */
223
+ const WIRE_VERSION = "1.0.0";
224
+ /**
225
+ * Stable hash of a rubric. Used to make scores comparable across runs:
226
+ * if the rubricVersion matches, the rubric was identical.
227
+ */
228
+ function hashRubric(rubric) {
229
+ const stable = stableStringify(rubric);
230
+ let h = 5381;
231
+ for (let i = 0; i < stable.length; i++) h = h * 33 ^ stable.charCodeAt(i);
232
+ return `${rubric.name}@${(h >>> 0).toString(16).padStart(8, "0")}`;
233
+ }
234
+ function stableStringify(value) {
235
+ if (Array.isArray(value)) return `[${value.map((item) => stableStringify(item)).join(",")}]`;
236
+ if (value && typeof value === "object") return `{${Object.entries(value).sort(([a], [b]) => a.localeCompare(b)).map(([key, item]) => `${JSON.stringify(key)}:${stableStringify(item)}`).join(",")}}`;
237
+ return JSON.stringify(value);
238
+ }
239
+ const BUILTIN_RUBRICS = { "anti-slop": {
240
+ name: "anti-slop",
241
+ description: "Voice and signal quality for content aimed at senior engineers. Catches AI cadence, marketing tone, and engagement-bait shapes.",
242
+ systemPrompt: `You are evaluating a piece of content written for senior engineers and technical founders.
243
+
244
+ You score three things:
245
+ - buyer_quality (0..1): would a senior engineer in the target ICP find this worth their attention? High = specific, earned, technically interesting. Low = generic, hyped, off-target.
246
+ - voice (0..1): does it read like a person who built the thing, or like AI/marketing copy?
247
+ - signal (0..1): does it contain a non-obvious detail, constraint, or claim a reader couldn't get from the public docs?
248
+
249
+ Detect failure modes (return ids matching):
250
+ - ai-cadence: rule-of-three openings, em-dash flourish, "Let me explain", "Here's the thing", AI rhythm
251
+ - marketing-tone: "We're excited to announce", "thrilled", "delighted", "game-changer", buzzword stack
252
+ - vague-claim: technical claim without a specific component, file, or measurement
253
+ - no-hook: opening doesn't earn attention from the target reader
254
+ - engagement-bait: "agree?", "thoughts?", listicles, controversy-fishing, hook-detail-pitch
255
+ - off-icp: content shape would attract motivational/grift/hype audiences instead of buyers
256
+ - stale-claim: repeats a positioning line we've used many times this month
257
+
258
+ Detect wins (return ids matching):
259
+ - specific-component: names a real file, component, or measurement
260
+ - earned-detail: shares a non-obvious detail not derivable from public docs
261
+ - constraint-articulated: names a real tradeoff and the side chosen
262
+ - honest-failure: describes a real failure mode and what was done about it
263
+
264
+ Return ONLY JSON matching the response schema. Be conservative — most content has 0-1 wins and 1-2 failure modes, not many of each.`,
265
+ dimensions: [
266
+ {
267
+ id: "buyer_quality",
268
+ description: "Would the target buyer find this worth attention?",
269
+ weight: .5,
270
+ min: 0,
271
+ max: 1
272
+ },
273
+ {
274
+ id: "voice",
275
+ description: "Does it sound like a builder, not AI or marketing?",
276
+ weight: .3,
277
+ min: 0,
278
+ max: 1
279
+ },
280
+ {
281
+ id: "signal",
282
+ description: "Non-obvious detail, constraint, or claim?",
283
+ weight: .2,
284
+ min: 0,
285
+ max: 1
286
+ }
287
+ ],
288
+ failureModes: [
289
+ {
290
+ id: "ai-cadence",
291
+ description: "AI-rhythm openings and transitions"
292
+ },
293
+ {
294
+ id: "marketing-tone",
295
+ description: "Buzzwords, hype, corporate-PR voice"
296
+ },
297
+ {
298
+ id: "vague-claim",
299
+ description: "Technical claim without specifics"
300
+ },
301
+ {
302
+ id: "no-hook",
303
+ description: "Opening fails to earn attention"
304
+ },
305
+ {
306
+ id: "engagement-bait",
307
+ description: "Listicle/controversy/agree-pattern"
308
+ },
309
+ {
310
+ id: "off-icp",
311
+ description: "Voice attracts the wrong audience"
312
+ },
313
+ {
314
+ id: "stale-claim",
315
+ description: "Reuses an over-used positioning line"
316
+ }
317
+ ],
318
+ wins: [
319
+ {
320
+ id: "specific-component",
321
+ description: "Names a real file/component/number"
322
+ },
323
+ {
324
+ id: "earned-detail",
325
+ description: "Detail not in public docs"
326
+ },
327
+ {
328
+ id: "constraint-articulated",
329
+ description: "Names a real tradeoff"
330
+ },
331
+ {
332
+ id: "honest-failure",
333
+ description: "Describes a real failure honestly"
334
+ }
335
+ ]
336
+ } };
337
+ /** Get a built-in rubric by name, or undefined. */
338
+ function getBuiltinRubric(name) {
339
+ return BUILTIN_RUBRICS[name];
340
+ }
341
+ /** List built-in rubrics with their stable versions. */
342
+ function listBuiltinRubrics() {
343
+ return Object.values(BUILTIN_RUBRICS).map((r) => ({
344
+ name: r.name,
345
+ description: r.description,
346
+ dimensions: r.dimensions.map((d) => ({
347
+ id: d.id,
348
+ description: d.description,
349
+ weight: d.weight
350
+ })),
351
+ failureModes: r.failureModes.map((f) => f.id),
352
+ rubricVersion: hashRubric(r)
353
+ }));
354
+ }
355
+ //#endregion
356
+ //#region src/wire/handlers.ts
357
+ /**
358
+ * Pure handler functions — the "business logic" behind every wire-protocol
359
+ * method. The HTTP server (`server.ts`) and the stdio RPC (`rpc.ts`) both
360
+ * call these. Tests call these directly without spinning a server.
361
+ *
362
+ * Each handler:
363
+ * - Takes a parsed request (already Zod-validated by the transport).
364
+ * - Returns a result that matches the response schema.
365
+ * - Throws `WireError` for caller-fixable errors (404, 400, 422).
366
+ * - Lets unexpected errors bubble — the transport maps them to 500.
367
+ */
368
+ /** Caller-fixable error. The transport renders this to 4xx + ErrorResponse. */
369
+ var WireError = class extends Error {
370
+ code;
371
+ status;
372
+ details;
373
+ constructor(code, message, status = 400, details) {
374
+ super(message);
375
+ this.code = code;
376
+ this.status = status;
377
+ this.details = details;
378
+ this.name = "WireError";
379
+ }
380
+ };
381
+ /** The JSON schema we ask the judging LLM to fill in. */
382
+ function judgeOutputSchema(rubric) {
383
+ return {
384
+ name: "JudgeOutput",
385
+ schema: {
386
+ type: "object",
387
+ additionalProperties: false,
388
+ properties: {
389
+ dimensions: {
390
+ type: "object",
391
+ additionalProperties: false,
392
+ properties: Object.fromEntries(rubric.dimensions.map((d) => [d.id, {
393
+ type: "number",
394
+ minimum: d.min,
395
+ maximum: d.max
396
+ }])),
397
+ required: rubric.dimensions.map((d) => d.id)
398
+ },
399
+ failureModes: {
400
+ type: "array",
401
+ items: {
402
+ type: "string",
403
+ enum: rubric.failureModes.map((f) => f.id)
404
+ }
405
+ },
406
+ wins: {
407
+ type: "array",
408
+ items: {
409
+ type: "string",
410
+ enum: rubric.wins.map((w) => w.id)
411
+ }
412
+ },
413
+ rationale: { type: "string" }
414
+ },
415
+ required: ["dimensions", "rationale"]
416
+ }
417
+ };
418
+ }
419
+ function validateJudgeOutput(value, rubric) {
420
+ if (!value || typeof value !== "object") throw new WireError("judge_error", "Judge returned malformed output.", 500, value);
421
+ const raw = value;
422
+ const rawDimensions = raw.dimensions;
423
+ if (!rawDimensions || typeof rawDimensions !== "object" || Array.isArray(rawDimensions)) throw new WireError("judge_error", "Judge returned malformed dimensions.", 500, value);
424
+ const dimensions = {};
425
+ const dimensionRecord = rawDimensions;
426
+ for (const dim of rubric.dimensions) {
427
+ const score = dimensionRecord[dim.id];
428
+ if (typeof score !== "number" || !Number.isFinite(score) || score < dim.min || score > dim.max) throw new WireError("judge_error", `Judge returned invalid score for dimension "${dim.id}".`, 500, value);
429
+ dimensions[dim.id] = score;
430
+ }
431
+ const allowedFailures = new Set(rubric.failureModes.map((mode) => mode.id));
432
+ const allowedWins = new Set(rubric.wins.map((win) => win.id));
433
+ const failureModes = validateIdArray(raw.failureModes, allowedFailures, "failureModes", value);
434
+ const wins = validateIdArray(raw.wins, allowedWins, "wins", value);
435
+ if (typeof raw.rationale !== "string" || raw.rationale.trim().length === 0) throw new WireError("judge_error", "Judge returned missing rationale.", 500, value);
436
+ return {
437
+ dimensions,
438
+ failureModes,
439
+ wins,
440
+ rationale: raw.rationale
441
+ };
442
+ }
443
+ function validateIdArray(raw, allowed, field, original) {
444
+ if (raw === void 0) return [];
445
+ if (!Array.isArray(raw)) throw new WireError("judge_error", `Judge returned non-array ${field}.`, 500, original);
446
+ const out = [];
447
+ for (const item of raw) {
448
+ if (typeof item !== "string" || !allowed.has(item)) throw new WireError("judge_error", `Judge returned unknown ${field} id "${String(item)}".`, 500, original);
449
+ out.push(item);
450
+ }
451
+ return out;
452
+ }
453
+ function compositeScore(dimensions, rubric) {
454
+ let weighted = 0;
455
+ let totalWeight = 0;
456
+ for (const dim of rubric.dimensions) {
457
+ const raw = dimensions[dim.id] ?? 0;
458
+ const range = dim.max - dim.min || 1;
459
+ const normalized = Math.max(0, Math.min(1, (raw - dim.min) / range));
460
+ weighted += normalized * dim.weight;
461
+ totalWeight += dim.weight;
462
+ }
463
+ return totalWeight > 0 ? weighted / totalWeight : 0;
464
+ }
465
+ function buildJudgePrompt(content, context) {
466
+ const ctx = context && Object.keys(context).length ? JSON.stringify(context) : "";
467
+ return [
468
+ `CONTENT TO JUDGE:`,
469
+ content,
470
+ "",
471
+ ctx ? `CONTEXT (metadata, analytics, etc.):` : "",
472
+ ctx ? ctx : ""
473
+ ].filter(Boolean).join("\n");
474
+ }
475
+ const DEFAULT_JUDGE_MODEL = "claude-sonnet-4-6";
476
+ async function handleJudge(req, options = {}) {
477
+ let rubric;
478
+ if (req.rubricName) {
479
+ const found = getBuiltinRubric(req.rubricName);
480
+ if (!found) throw new WireError("rubric_not_found", `No built-in rubric named "${req.rubricName}".`, 404);
481
+ rubric = found;
482
+ } else if (req.rubric) rubric = req.rubric;
483
+ else throw new WireError("validation_error", "Provide either `rubricName` or `rubric`.", 422);
484
+ if (options.routeRequirements) try {
485
+ assertLlmRoute(options.llm ?? {}, options.routeRequirements);
486
+ } catch (error) {
487
+ if (!(error instanceof LlmRouteAssertionError)) throw error;
488
+ const noEndpoint = error.reason === "no_explicit_base_url";
489
+ throw new WireError(noEndpoint ? "llm_not_configured" : "llm_route_rejected", noEndpoint ? "No model endpoint is configured. Pass llm.baseUrl or configure the CLI provider environment variables." : error.message, 503, { reason: error.reason });
490
+ }
491
+ const startedAt = Date.now();
492
+ const model = req.model ?? options.defaultModel ?? DEFAULT_JUDGE_MODEL;
493
+ const request = {
494
+ model,
495
+ messages: [{
496
+ role: "system",
497
+ content: rubric.systemPrompt
498
+ }, {
499
+ role: "user",
500
+ content: buildJudgePrompt(req.content, req.context)
501
+ }],
502
+ jsonSchema: judgeOutputSchema(rubric),
503
+ temperature: 0,
504
+ maxTokens: 4e3,
505
+ timeoutMs: 6e4
506
+ };
507
+ const paid = await (options.costLedger ?? new CostLedger()).runPaidCall({
508
+ channel: "judge",
509
+ phase: options.costPhase ?? "wire.judge",
510
+ actor: `wire.${req.rubricName ?? "inline"}`,
511
+ model,
512
+ maximumCharge: maximumChargeForLlmRequest(request, options.llm),
513
+ signal: options.signal,
514
+ execute: (signal, callId) => callLlmJson(request, {
515
+ ...options.llm,
516
+ signal,
517
+ idempotencyKey: callId
518
+ }),
519
+ receipt: ({ result }) => costReceiptFromLlm(result),
520
+ receiptFromError: costReceiptFromLlmError
521
+ });
522
+ if (!paid.succeeded) throw paid.error;
523
+ const { value, result } = paid.value;
524
+ const output = validateJudgeOutput(value, rubric);
525
+ const composite = compositeScore(output.dimensions, rubric);
526
+ const durationMs = Date.now() - startedAt;
527
+ return {
528
+ composite,
529
+ dimensions: output.dimensions,
530
+ failureModes: output.failureModes ?? [],
531
+ wins: output.wins ?? [],
532
+ rationale: output.rationale,
533
+ rubricVersion: hashRubric(rubric),
534
+ model: result.model,
535
+ durationMs
536
+ };
537
+ }
538
+ function handleListRubrics() {
539
+ return { rubrics: listBuiltinRubrics() };
540
+ }
541
+ let CACHED_VERSION;
542
+ function readPackageVersion() {
543
+ if (CACHED_VERSION) return CACHED_VERSION;
544
+ const here = dirname(fileURLToPath(import.meta.url));
545
+ const candidates = [resolve(here, "..", "..", "package.json"), resolve(here, "..", "package.json")];
546
+ for (const path of candidates) try {
547
+ const pkg = JSON.parse(readFileSync(path, "utf-8"));
548
+ if (pkg.version) {
549
+ CACHED_VERSION = pkg.version;
550
+ return pkg.version;
551
+ }
552
+ } catch {}
553
+ return "0.0.0-unknown";
554
+ }
555
+ function handleVersion() {
556
+ return {
557
+ package: "@tangle-network/agent-eval",
558
+ version: readPackageVersion(),
559
+ wireVersion: WIRE_VERSION,
560
+ apiSurface: [
561
+ "judge",
562
+ "listRubrics",
563
+ "version",
564
+ "feedback.ingest",
565
+ "traces.ingest"
566
+ ]
567
+ };
568
+ }
569
+ /**
570
+ * `POST /v1/traces/ingest` — accept a batch of `TraceEvent`s from the
571
+ * production runtime. Best-effort: each event is appended independently;
572
+ * one bad event does not poison the batch.
573
+ *
574
+ * Idempotency: the underlying store is append-only; consumers retrying
575
+ * the same payload will get duplicate events. Consumers should
576
+ * de-duplicate by `eventId` downstream — production traces frequently
577
+ * land via at-least-once buses (Kafka, SQS) where dedup is unavoidable.
578
+ */
579
+ async function handleTracesIngest(req, stores) {
580
+ if (!stores.traceStore) throw new WireError("service_unavailable", "No trace store configured on this server. Pass `traceStore` to `createApp`.", 503);
581
+ const errors = [];
582
+ let accepted = 0;
583
+ for (const event of req.events) try {
584
+ await stores.traceStore.appendEvent(event);
585
+ accepted++;
586
+ } catch (err) {
587
+ errors.push({
588
+ eventId: event.eventId,
589
+ message: err instanceof Error ? err.message : String(err)
590
+ });
591
+ }
592
+ return {
593
+ accepted,
594
+ rejected: errors.length,
595
+ errors
596
+ };
597
+ }
598
+ /**
599
+ * `POST /v1/feedback` — accept a single `FeedbackTrajectory` from the
600
+ * production runtime. Idempotent on `id`: re-posting the same trajectory
601
+ * replaces the prior record.
602
+ */
603
+ async function handleFeedbackIngest(req, stores) {
604
+ if (!stores.feedbackStore) throw new WireError("service_unavailable", "No feedback store configured on this server. Pass `feedbackStore` to `createApp`.", 503);
605
+ await stores.feedbackStore.save(req);
606
+ return {
607
+ id: req.id,
608
+ persisted: true
609
+ };
610
+ }
611
+ //#endregion
612
+ //#region src/wire/openapi.ts
613
+ /**
614
+ * Build an OpenAPI spec from the wire schemas.
615
+ *
616
+ * The spec is the contract that other-language clients (Python, Rust,
617
+ * Go, …) generate from. There is no hand-written client — clients are
618
+ * derived artifacts of this file plus `schemas.ts`.
619
+ *
620
+ * Run `pnpm openapi` (defined in package.json) to write the spec to
621
+ * `dist/openapi.json`. CI uses that file to regenerate the Python
622
+ * client and gate the dual-publish workflow.
623
+ */
624
+ function buildOpenApi(packageVersion) {
625
+ const registry = new OpenAPIRegistry();
626
+ registry.register("JudgeRequest", JudgeRequestSchema);
627
+ registry.register("JudgeResult", JudgeResultSchema);
628
+ registry.register("ListRubricsResponse", ListRubricsResponseSchema);
629
+ registry.register("VersionResponse", VersionResponseSchema);
630
+ registry.register("HealthResponse", HealthResponseSchema);
631
+ registry.register("ErrorResponse", ErrorResponseSchema);
632
+ registry.register("TracesIngestRequest", TracesIngestRequestSchema);
633
+ registry.register("TracesIngestResponse", TracesIngestResponseSchema);
634
+ registry.register("FeedbackTrajectory", FeedbackTrajectorySchema);
635
+ registry.register("FeedbackIngestResponse", FeedbackIngestResponseSchema);
636
+ registry.registerPath({
637
+ method: "post",
638
+ path: "/v1/judge",
639
+ summary: "Score a piece of content against a rubric",
640
+ description: "Runs the judging LLM with the named (or inline) rubric and returns dimension scores, detected failure modes, wins, and a composite score in 0..1.",
641
+ request: { body: { content: { "application/json": { schema: JudgeRequestSchema } } } },
642
+ responses: {
643
+ 200: {
644
+ description: "Successful judgement",
645
+ content: { "application/json": { schema: JudgeResultSchema } }
646
+ },
647
+ 400: {
648
+ description: "Validation error",
649
+ content: { "application/json": { schema: ErrorResponseSchema } }
650
+ },
651
+ 404: {
652
+ description: "Rubric not found",
653
+ content: { "application/json": { schema: ErrorResponseSchema } }
654
+ },
655
+ 500: {
656
+ description: "Judge error",
657
+ content: { "application/json": { schema: ErrorResponseSchema } }
658
+ }
659
+ }
660
+ });
661
+ registry.registerPath({
662
+ method: "get",
663
+ path: "/v1/rubrics",
664
+ summary: "List built-in rubrics",
665
+ description: "Returns every rubric registered server-side, with their dimensions and stable rubricVersion hash.",
666
+ responses: { 200: {
667
+ description: "Listing",
668
+ content: { "application/json": { schema: ListRubricsResponseSchema } }
669
+ } }
670
+ });
671
+ registry.registerPath({
672
+ method: "get",
673
+ path: "/v1/version",
674
+ summary: "Server and wire-protocol version",
675
+ description: "Match your client version to `version`; check `wireVersion` for compatibility.",
676
+ responses: { 200: {
677
+ description: "Version info",
678
+ content: { "application/json": { schema: VersionResponseSchema } }
679
+ } }
680
+ });
681
+ registry.registerPath({
682
+ method: "get",
683
+ path: "/healthz",
684
+ summary: "Liveness check",
685
+ responses: { 200: {
686
+ description: "OK",
687
+ content: { "application/json": { schema: HealthResponseSchema } }
688
+ } }
689
+ });
690
+ registry.registerPath({
691
+ method: "post",
692
+ path: "/v1/traces/ingest",
693
+ summary: "Ingest a batch of production TraceEvents",
694
+ description: "Append a batch of TraceEvents to the configured TraceStore. Accepts application/json ({events:[...]}) or application/x-ndjson (one event per line). Returns counts of accepted + rejected events.",
695
+ request: { body: { content: {
696
+ "application/json": { schema: TracesIngestRequestSchema },
697
+ "application/x-ndjson": { schema: TracesIngestRequestSchema }
698
+ } } },
699
+ responses: {
700
+ 200: {
701
+ description: "Ingestion summary",
702
+ content: { "application/json": { schema: TracesIngestResponseSchema } }
703
+ },
704
+ 400: {
705
+ description: "Validation error",
706
+ content: { "application/json": { schema: ErrorResponseSchema } }
707
+ },
708
+ 401: {
709
+ description: "Unauthorized (when bearer auth is configured)",
710
+ content: { "application/json": { schema: ErrorResponseSchema } }
711
+ },
712
+ 503: {
713
+ description: "No trace store configured",
714
+ content: { "application/json": { schema: ErrorResponseSchema } }
715
+ }
716
+ }
717
+ });
718
+ registry.registerPath({
719
+ method: "post",
720
+ path: "/v1/feedback",
721
+ summary: "Ingest a FeedbackTrajectory from production",
722
+ description: "Persist a single FeedbackTrajectory. Idempotent on trajectory.id — re-posting replaces the prior record. Used by production runtimes to forward user 👍/👎/edits into the eval substrate.",
723
+ request: { body: { content: { "application/json": { schema: FeedbackTrajectorySchema } } } },
724
+ responses: {
725
+ 200: {
726
+ description: "Persisted",
727
+ content: { "application/json": { schema: FeedbackIngestResponseSchema } }
728
+ },
729
+ 400: {
730
+ description: "Validation error",
731
+ content: { "application/json": { schema: ErrorResponseSchema } }
732
+ },
733
+ 401: {
734
+ description: "Unauthorized (when bearer auth is configured)",
735
+ content: { "application/json": { schema: ErrorResponseSchema } }
736
+ },
737
+ 503: {
738
+ description: "No feedback store configured",
739
+ content: { "application/json": { schema: ErrorResponseSchema } }
740
+ }
741
+ }
742
+ });
743
+ const doc = new OpenApiGeneratorV31(registry.definitions).generateDocument({
744
+ openapi: "3.1.0",
745
+ info: {
746
+ title: "@tangle-network/agent-eval — wire protocol",
747
+ version: packageVersion,
748
+ description: `HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.
749
+
750
+ Wire-protocol version: ${WIRE_VERSION}. Bumps on breaking changes to request/response schemas.`,
751
+ contact: {
752
+ name: "Tangle Network",
753
+ url: "https://github.com/tangle-network/agent-eval"
754
+ },
755
+ license: { name: "MIT" }
756
+ },
757
+ servers: [{
758
+ url: "http://localhost:5005",
759
+ description: "Local agent-eval serve"
760
+ }]
761
+ });
762
+ const rubricRef = { $ref: "#/components/schemas/Rubric" };
763
+ const commonJudgeFields = {
764
+ content: {
765
+ type: "string",
766
+ minLength: 1
767
+ },
768
+ context: {
769
+ type: "object",
770
+ additionalProperties: true
771
+ },
772
+ model: { type: "string" }
773
+ };
774
+ doc.components ??= {};
775
+ doc.components.schemas ??= {};
776
+ doc.components.schemas.JudgeRequest = {
777
+ oneOf: [{
778
+ type: "object",
779
+ additionalProperties: false,
780
+ required: ["rubricName", "content"],
781
+ properties: {
782
+ rubricName: {
783
+ type: "string",
784
+ minLength: 1
785
+ },
786
+ ...commonJudgeFields
787
+ }
788
+ }, {
789
+ type: "object",
790
+ additionalProperties: false,
791
+ required: ["rubric", "content"],
792
+ properties: {
793
+ rubric: rubricRef,
794
+ ...commonJudgeFields
795
+ }
796
+ }],
797
+ description: "Judge request. Provide exactly one of rubricName or rubric."
798
+ };
799
+ return doc;
800
+ }
801
+ //#endregion
802
+ //#region src/wire/rpc.ts
803
+ async function dispatchRpc(req, options = {}) {
804
+ try {
805
+ switch (req.method) {
806
+ case "judge": {
807
+ const parsed = JudgeRequestSchema.safeParse(req.params);
808
+ if (!parsed.success) return { error: {
809
+ code: "validation_error",
810
+ message: "params did not match JudgeRequest schema.",
811
+ details: parsed.error.issues
812
+ } };
813
+ return { result: await handleJudge(parsed.data, {
814
+ llm: options.llm,
815
+ defaultModel: options.judgeModel,
816
+ routeRequirements: options.llmRouteRequirements
817
+ }) };
818
+ }
819
+ case "listRubrics": return { result: handleListRubrics() };
820
+ case "version": return { result: handleVersion() };
821
+ default: return { error: {
822
+ code: "unknown_method",
823
+ message: `No such method: ${req.method}`
824
+ } };
825
+ }
826
+ } catch (err) {
827
+ if (err instanceof WireError) return { error: {
828
+ code: err.code,
829
+ message: err.message,
830
+ details: err.details
831
+ } };
832
+ return { error: {
833
+ code: "internal_error",
834
+ message: err instanceof Error ? err.message : String(err)
835
+ } };
836
+ }
837
+ }
838
+ async function readAll(stream) {
839
+ const chunks = [];
840
+ for await (const chunk of stream) chunks.push(Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk));
841
+ return Buffer.concat(chunks).toString("utf-8");
842
+ }
843
+ /** Read one JSON request from stdin, write one JSON response to stdout. */
844
+ async function runRpcOnce(method, options = {}) {
845
+ const raw = await readAll(process.stdin);
846
+ let req;
847
+ try {
848
+ const body = JSON.parse(raw);
849
+ req = method ? {
850
+ method,
851
+ params: body
852
+ } : body;
853
+ } catch (err) {
854
+ process.stdout.write(`${JSON.stringify({ error: {
855
+ code: "parse_error",
856
+ message: `stdin was not valid JSON: ${err instanceof Error ? err.message : String(err)}`
857
+ } })}\n`);
858
+ return 1;
859
+ }
860
+ const out = await dispatchRpc(req, options);
861
+ process.stdout.write(`${JSON.stringify(out)}\n`);
862
+ return "error" in out ? 1 : 0;
863
+ }
864
+ /** Read JSONL requests from stdin, write JSONL responses to stdout. */
865
+ async function runRpcBatch(method, options = {}) {
866
+ const lines = (await readAll(process.stdin)).split("\n").filter((l) => l.trim().length > 0);
867
+ let exitCode = 0;
868
+ for (const line of lines) {
869
+ let req;
870
+ try {
871
+ const body = JSON.parse(line);
872
+ req = method ? {
873
+ method,
874
+ params: body
875
+ } : body;
876
+ } catch (err) {
877
+ process.stdout.write(`${JSON.stringify({ error: {
878
+ code: "parse_error",
879
+ message: `line was not valid JSON: ${err instanceof Error ? err.message : String(err)}`
880
+ } })}\n`);
881
+ exitCode = 1;
882
+ continue;
883
+ }
884
+ const out = await dispatchRpc(req, options);
885
+ process.stdout.write(`${JSON.stringify(out)}\n`);
886
+ if ("error" in out) exitCode = 1;
887
+ }
888
+ return exitCode;
889
+ }
890
+ //#endregion
891
+ //#region src/wire/server.ts
892
+ /**
893
+ * HTTP transport for the wire protocol.
894
+ *
895
+ * Hono + @hono/node-server. Every endpoint:
896
+ * 1. Validates the request against its Zod schema.
897
+ * 2. Calls the matching handler in `handlers.ts`.
898
+ * 3. Renders 4xx for `WireError` with structured body, 500 for unexpected.
899
+ *
900
+ * The server holds optional `IngestionStores` (passed to `createApp`)
901
+ * to receive production traces and user feedback. With no stores wired,
902
+ * the ingestion endpoints return 503 — read endpoints (`/v1/judge`,
903
+ * `/v1/rubrics`, `/v1/version`) remain fully functional.
904
+ *
905
+ * Run via `agent-eval serve --port 5005`.
906
+ */
907
+ const STARTED_AT = Date.now();
908
+ const AUTH_EXEMPT_PATHS = /* @__PURE__ */ new Set([
909
+ "/healthz",
910
+ "/v1/version",
911
+ "/openapi.json"
912
+ ]);
913
+ function createApp(opts = {}) {
914
+ const app = new Hono();
915
+ app.use("*", cors());
916
+ if (opts.auth) {
917
+ const verify = opts.auth.bearer;
918
+ app.use("*", async (c, next) => {
919
+ const path = new URL(c.req.url).pathname;
920
+ if (AUTH_EXEMPT_PATHS.has(path)) return next();
921
+ const match = (c.req.header("authorization") ?? "").match(/^Bearer\s+(.+)$/i);
922
+ if (!match) throw new WireError("unauthorized", "Missing or malformed Authorization header.", 401);
923
+ const token = match[1];
924
+ if (!(typeof verify === "string" ? token === verify : await verify(token))) throw new WireError("unauthorized", "Invalid bearer token.", 401);
925
+ return next();
926
+ });
927
+ }
928
+ app.onError((err, c) => {
929
+ if (err instanceof WireError) {
930
+ const status = err.status;
931
+ return c.json({ error: {
932
+ code: err.code,
933
+ message: err.message,
934
+ details: err.details
935
+ } }, status);
936
+ }
937
+ console.error("[agent-eval] unhandled error:", err);
938
+ return c.json({ error: {
939
+ code: "internal_error",
940
+ message: "Internal server error."
941
+ } }, 500);
942
+ });
943
+ app.get("/healthz", (c) => c.json({
944
+ status: "ok",
945
+ uptimeSec: (Date.now() - STARTED_AT) / 1e3
946
+ }));
947
+ app.get("/v1/version", (c) => c.json(handleVersion()));
948
+ app.get("/v1/rubrics", (c) => c.json(handleListRubrics()));
949
+ app.post("/v1/judge", async (c) => {
950
+ const raw = await c.req.json().catch(() => null);
951
+ if (raw == null) throw new WireError("validation_error", "Request body must be JSON.", 400);
952
+ const parsed = JudgeRequestSchema.safeParse(raw);
953
+ if (!parsed.success) throw new WireError("validation_error", "Request did not match JudgeRequest schema.", 400, parsed.error.issues);
954
+ const result = await handleJudge(parsed.data, {
955
+ llm: opts.llm,
956
+ defaultModel: opts.judgeModel,
957
+ routeRequirements: opts.llmRouteRequirements
958
+ });
959
+ return c.json(result);
960
+ });
961
+ app.post("/v1/traces/ingest", async (c) => {
962
+ const contentType = c.req.header("content-type") ?? "";
963
+ let payload;
964
+ if (contentType.includes("application/x-ndjson")) payload = { events: (await c.req.text()).split("\n").map((line) => line.trim()).filter((line) => line.length > 0).map((line) => {
965
+ try {
966
+ return JSON.parse(line);
967
+ } catch {
968
+ throw new WireError("validation_error", "NDJSON line did not parse as JSON.", 400, line.slice(0, 200));
969
+ }
970
+ }) };
971
+ else payload = await c.req.json().catch(() => null);
972
+ if (payload == null) throw new WireError("validation_error", "Request body must be JSON or NDJSON.", 400);
973
+ const parsed = TracesIngestRequestSchema.safeParse(payload);
974
+ if (!parsed.success) throw new WireError("validation_error", "Request did not match TracesIngestRequest schema.", 400, parsed.error.issues);
975
+ const result = await handleTracesIngest(parsed.data, opts.stores ?? {});
976
+ return c.json(result);
977
+ });
978
+ app.post("/v1/feedback", async (c) => {
979
+ const raw = await c.req.json().catch(() => null);
980
+ if (raw == null) throw new WireError("validation_error", "Request body must be JSON.", 400);
981
+ const parsed = FeedbackTrajectorySchema.safeParse(raw);
982
+ if (!parsed.success) throw new WireError("validation_error", "Request did not match FeedbackTrajectory schema.", 400, parsed.error.issues);
983
+ const result = await handleFeedbackIngest(parsed.data, opts.stores ?? {});
984
+ return c.json(result);
985
+ });
986
+ app.get("/openapi.json", (c) => c.json(buildOpenApi(handleVersion().version)));
987
+ return app;
988
+ }
989
+ function startServer(opts = {}) {
990
+ const app = createApp(opts);
991
+ const port = opts.port ?? 5005;
992
+ const host = opts.host ?? "127.0.0.1";
993
+ return serve({
994
+ fetch: app.fetch,
995
+ port,
996
+ hostname: host
997
+ }, ({ address, port: actualPort }) => {
998
+ console.log(`[agent-eval] serving on http://${address}:${actualPort}`);
999
+ });
1000
+ }
1001
+ /**
1002
+ * Promise-returning variant of `startServer` that resolves once the server is
1003
+ * listening and surfaces the resolved bound port. Use this from smoke tests
1004
+ * (`startServerAsync({ port: 0 })`) and any caller that needs to dial back.
1005
+ */
1006
+ function startServerAsync(opts = {}) {
1007
+ const app = createApp(opts);
1008
+ const port = opts.port ?? 5005;
1009
+ const host = opts.host ?? "127.0.0.1";
1010
+ return new Promise((resolve, reject) => {
1011
+ let settled = false;
1012
+ let server;
1013
+ server = serve({
1014
+ fetch: app.fetch,
1015
+ port,
1016
+ hostname: host
1017
+ }, ({ address, port: actualPort }) => {
1018
+ if (settled) return;
1019
+ settled = true;
1020
+ console.log(`[agent-eval] serving on http://${address}:${actualPort}`);
1021
+ resolve({
1022
+ server,
1023
+ port: actualPort,
1024
+ host: address,
1025
+ close: () => new Promise((res, rej) => {
1026
+ server.close((err) => err ? rej(err) : res());
1027
+ })
1028
+ });
1029
+ });
1030
+ server.on("error", (err) => {
1031
+ if (settled) return;
1032
+ settled = true;
1033
+ reject(err);
1034
+ });
1035
+ });
1036
+ }
1037
+ //#endregion
1038
+ export { TraceEventSchema as A, HealthResponseSchema as C, RubricDimensionSchema as D, ListRubricsResponseSchema as E, hashRubric as F, TracesIngestResponseSchema as M, VersionResponseSchema as N, RubricInfoSchema as O, WIRE_VERSION as P, FeedbackTrajectorySchema as S, JudgeResultSchema as T, ErrorResponseSchema as _, runRpcBatch as a, FeedbackIngestResponseSchema as b, WireError as c, handleListRubrics as d, handleTracesIngest as f, listBuiltinRubrics as g, getBuiltinRubric as h, dispatchRpc as i, TracesIngestRequestSchema as j, RubricSchema as k, handleFeedbackIngest as l, BUILTIN_RUBRICS as m, startServer as n, runRpcOnce as o, handleVersion as p, startServerAsync as r, buildOpenApi as s, createApp as t, handleJudge as u, FailureModeSchema as v, JudgeRequestSchema as w, FeedbackLabelSchema as x, FeedbackAttemptSchema as y };
1039
+
1040
+ //# sourceMappingURL=server-m5D9cvnG.js.map